{"timestamp_utc": "2026-04-11T19:23:21Z", "mode": "train", "global_step": 1, "epoch": 3.861600247142416e-05, "loss": -0.0412, "grad_norm": 12.506416320800781, "learning_rate": 1e-05, "num_tokens": 2187.0, "completions/mean_length": 86.375, "completions/min_length": 50.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.5479353666305542, "rewards/meter/std": 0.42796453833580017, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.46589815616607666, "rewards/total_composite/std": 0.35266923904418945, "reward": 0.46589815616607666, "reward_std": 0.35266923904418945, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20873695611953735, "sampling/sampling_logp_difference/max": 1.5729236602783203, "sampling/importance_sampling_ratio/min": 0.20743781328201294, "sampling/importance_sampling_ratio/mean": 1.005318284034729, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9424520283937454, "clip_ratio/low_mean": 0.05967577267438173, "clip_ratio/low_min": 0.05967577267438173, "clip_ratio/high_mean": 0.1411083247512579, "clip_ratio/high_max": 0.1411083247512579, "clip_ratio/region_mean": 0.20078409742563963, "reward_total_mean": 0.46589815616607666, "reward_meter_mean": 0.5479353666305542, "reward_meter_std": 0.42796453833580017, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.46589815616607666, "reward_total_composite_std": 0.35266923904418945, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1.0} {"timestamp_utc": "2026-04-11T19:23:26Z", "mode": "train", "global_step": 2, "epoch": 7.723200494284832e-05, "loss": 0.0621, "grad_norm": 17.5363826751709, "learning_rate": 9.996969696969698e-06, "num_tokens": 4001.0, "completions/mean_length": 57.75, "completions/min_length": 49.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.6565604209899902, "rewards/meter/std": 0.3700096607208252, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6565604209899902, "rewards/total_composite/std": 0.3700096607208252, "reward": 0.6565604209899902, "reward_std": 0.3700096607208252, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16318750381469727, "sampling/sampling_logp_difference/max": 1.986765742301941, "sampling/importance_sampling_ratio/min": 0.1371382474899292, "sampling/importance_sampling_ratio/mean": 1.0128601789474487, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9319793060421944, "clip_ratio/low_mean": 0.10001249238848686, "clip_ratio/low_min": 0.10001249238848686, "clip_ratio/high_mean": 0.05052456725388765, "clip_ratio/high_max": 0.05052456725388765, "clip_ratio/region_mean": 0.15053705964237452, "reward_total_mean": 0.6565604209899902, "reward_meter_mean": 0.6565604209899902, "reward_meter_std": 0.3700096607208252, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6565604209899902, "reward_total_composite_std": 0.3700096607208252, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 2.0} {"timestamp_utc": "2026-04-11T19:23:30Z", "mode": "train", "global_step": 3, "epoch": 0.00011584800741427248, "loss": 0.0587, "grad_norm": 17.113204956054688, "learning_rate": 9.993939393939395e-06, "num_tokens": 5663.0, "completions/mean_length": 41.75, "completions/min_length": 35.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.5669770240783691, "rewards/meter/std": 0.42305904626846313, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5669770240783691, "rewards/total_composite/std": 0.42305904626846313, "reward": 0.5669770240783691, "reward_std": 0.42305904626846313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1992362141609192, "sampling/sampling_logp_difference/max": 1.4587993621826172, "sampling/importance_sampling_ratio/min": 0.23251527547836304, "sampling/importance_sampling_ratio/mean": 1.04336678981781, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6740138679742813, "clip_ratio/low_mean": 0.11367196403443813, "clip_ratio/low_min": 0.11367196403443813, "clip_ratio/high_mean": 0.10511028952896595, "clip_ratio/high_max": 0.10511028952896595, "clip_ratio/region_mean": 0.21878225356340408, "reward_total_mean": 0.5669770240783691, "reward_meter_mean": 0.5669770240783691, "reward_meter_std": 0.42305904626846313, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5669770240783691, "reward_total_composite_std": 0.42305904626846313, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 3.0} {"timestamp_utc": "2026-04-11T19:23:34Z", "mode": "train", "global_step": 4, "epoch": 0.00015446400988569664, "loss": -0.1025, "grad_norm": 19.642213821411133, "learning_rate": 9.990909090909093e-06, "num_tokens": 7109.0, "completions/mean_length": 26.75, "completions/min_length": 17.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.75, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.856600284576416, "rewards/meter/std": 0.34479716420173645, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.856600284576416, "rewards/total_composite/std": 0.34479716420173645, "reward": 0.856600284576416, "reward_std": 0.34479716420173645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18150335550308228, "sampling/sampling_logp_difference/max": 1.3495612144470215, "sampling/importance_sampling_ratio/min": 0.2593540549278259, "sampling/importance_sampling_ratio/mean": 1.0328437089920044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2625475600361824, "clip_ratio/low_mean": 0.029411764815449715, "clip_ratio/low_min": 0.029411764815449715, "clip_ratio/high_mean": 0.15018654288724065, "clip_ratio/high_max": 0.15018654288724065, "clip_ratio/region_mean": 0.17959830770269036, "reward_total_mean": 0.856600284576416, "reward_meter_mean": 0.856600284576416, "reward_meter_std": 0.34479716420173645, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.856600284576416, "reward_total_composite_std": 0.34479716420173645, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 4.0} {"timestamp_utc": "2026-04-11T19:23:41Z", "mode": "train", "global_step": 5, "epoch": 0.0001930800123571208, "loss": -0.0437, "grad_norm": 9.840316772460938, "learning_rate": 9.987878787878788e-06, "num_tokens": 10053.0, "completions/mean_length": 163.0, "completions/min_length": 105.0, "completions/max_length": 227.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 163.0, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 227.0, "rewards/meter/mean": 0.4093240201473236, "rewards/meter/std": 0.3055945336818695, "rewards/count_adherence/mean": 0.9642857313156128, "rewards/count_adherence/std": 0.06613000482320786, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.2891073226928711, "rewards/total_composite/std": 0.30707788467407227, "reward": 0.2891073226928711, "reward_std": 0.30707788467407227, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2513454854488373, "sampling/sampling_logp_difference/max": 2.3320798873901367, "sampling/importance_sampling_ratio/min": 0.0970935970544815, "sampling/importance_sampling_ratio/mean": 1.0441721677780151, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7085169553756714, "clip_ratio/low_mean": 0.15030906535685062, "clip_ratio/low_min": 0.15030906535685062, "clip_ratio/high_mean": 0.08046106435358524, "clip_ratio/high_max": 0.08046106435358524, "clip_ratio/region_mean": 0.23077012971043587, "reward_total_mean": 0.2891073226928711, "reward_meter_mean": 0.4093240201473236, "reward_meter_std": 0.3055945336818695, "reward_count_adherence_mean": 0.9642857313156128, "reward_count_adherence_std": 0.06613000482320786, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.2891073226928711, "reward_total_composite_std": 0.30707788467407227, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 5.0} {"timestamp_utc": "2026-04-11T19:23:46Z", "mode": "train", "global_step": 6, "epoch": 0.00023169601482854495, "loss": -0.0554, "grad_norm": 12.211259841918945, "learning_rate": 9.984848484848485e-06, "num_tokens": 12078.0, "completions/mean_length": 95.125, "completions/min_length": 66.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.5617634057998657, "rewards/meter/std": 0.41573381423950195, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5588854551315308, "rewards/total_composite/std": 0.42005324363708496, "reward": 0.5588854551315308, "reward_std": 0.42005324363708496, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23301345109939575, "sampling/sampling_logp_difference/max": 1.7306222915649414, "sampling/importance_sampling_ratio/min": 0.17717412114143372, "sampling/importance_sampling_ratio/mean": 1.0142629146575928, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5477894991636276, "clip_ratio/low_mean": 0.1318096686154604, "clip_ratio/low_min": 0.1318096686154604, "clip_ratio/high_mean": 0.11287152394652367, "clip_ratio/high_max": 0.11287152394652367, "clip_ratio/region_mean": 0.24468119256198406, "reward_total_mean": 0.5588854551315308, "reward_meter_mean": 0.5617634057998657, "reward_meter_std": 0.41573381423950195, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5588854551315308, "reward_total_composite_std": 0.42005324363708496, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 6.0} {"timestamp_utc": "2026-04-11T19:23:51Z", "mode": "train", "global_step": 7, "epoch": 0.0002703120172999691, "loss": 0.2296, "grad_norm": 15.44117546081543, "learning_rate": 9.981818181818183e-06, "num_tokens": 14125.0, "completions/mean_length": 84.875, "completions/min_length": 63.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.875, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.6451665163040161, "rewards/meter/std": 0.29868313670158386, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6451665163040161, "rewards/total_composite/std": 0.29868313670158386, "reward": 0.6451665163040161, "reward_std": 0.29868316650390625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2560131847858429, "sampling/sampling_logp_difference/max": 1.7976973056793213, "sampling/importance_sampling_ratio/min": 0.1656799614429474, "sampling/importance_sampling_ratio/mean": 1.0209758281707764, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2808013558387756, "clip_ratio/low_mean": 0.10475845448672771, "clip_ratio/low_min": 0.10475845448672771, "clip_ratio/high_mean": 0.12872153520584106, "clip_ratio/high_max": 0.12872153520584106, "clip_ratio/region_mean": 0.23347998969256878, "reward_total_mean": 0.6451665163040161, "reward_meter_mean": 0.6451665163040161, "reward_meter_std": 0.29868313670158386, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6451665163040161, "reward_total_composite_std": 0.29868313670158386, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 7.0} {"timestamp_utc": "2026-04-11T19:23:56Z", "mode": "train", "global_step": 8, "epoch": 0.00030892801977139327, "loss": 0.1614, "grad_norm": 17.118240356445312, "learning_rate": 9.97878787878788e-06, "num_tokens": 15903.0, "completions/mean_length": 56.25, "completions/min_length": 39.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8440761566162109, "rewards/meter/std": 0.24688278138637543, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8440761566162109, "rewards/total_composite/std": 0.24688278138637543, "reward": 0.8440761566162109, "reward_std": 0.24688279628753662, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22910818457603455, "sampling/sampling_logp_difference/max": 1.7483234405517578, "sampling/importance_sampling_ratio/min": 0.1740655153989792, "sampling/importance_sampling_ratio/mean": 1.0035207271575928, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.992170736193657, "clip_ratio/low_mean": 0.03434433601796627, "clip_ratio/low_min": 0.03434433601796627, "clip_ratio/high_mean": 0.19223029538989067, "clip_ratio/high_max": 0.19223029538989067, "clip_ratio/region_mean": 0.22657463140785694, "reward_total_mean": 0.8440761566162109, "reward_meter_mean": 0.8440761566162109, "reward_meter_std": 0.24688278138637543, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8440761566162109, "reward_total_composite_std": 0.24688278138637543, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 8.0} {"timestamp_utc": "2026-04-11T19:24:00Z", "mode": "train", "global_step": 9, "epoch": 0.00034754402224281743, "loss": 0.0706, "grad_norm": 16.50942039489746, "learning_rate": 9.975757575757577e-06, "num_tokens": 17535.0, "completions/mean_length": 47.0, "completions/min_length": 39.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.0, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.6138224601745605, "rewards/meter/std": 0.3940174877643585, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6138224601745605, "rewards/total_composite/std": 0.3940174877643585, "reward": 0.6138224601745605, "reward_std": 0.3940175175666809, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20549029111862183, "sampling/sampling_logp_difference/max": 2.07395076751709, "sampling/importance_sampling_ratio/min": 0.12568823993206024, "sampling/importance_sampling_ratio/mean": 1.0077967643737793, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6416602730751038, "clip_ratio/low_mean": 0.07851519016548991, "clip_ratio/low_min": 0.07851519016548991, "clip_ratio/high_mean": 0.10056564025580883, "clip_ratio/high_max": 0.10056564025580883, "clip_ratio/region_mean": 0.17908083042129874, "reward_total_mean": 0.6138224601745605, "reward_meter_mean": 0.6138224601745605, "reward_meter_std": 0.3940174877643585, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6138224601745605, "reward_total_composite_std": 0.3940174877643585, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 9.0} {"timestamp_utc": "2026-04-11T19:24:06Z", "mode": "train", "global_step": 10, "epoch": 0.0003861600247142416, "loss": 0.2066, "grad_norm": 18.21146583557129, "learning_rate": 9.972727272727274e-06, "num_tokens": 19339.0, "completions/mean_length": 56.5, "completions/min_length": 38.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.8475043773651123, "rewards/meter/std": 0.340963751077652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7867398858070374, "rewards/total_composite/std": 0.35842904448509216, "reward": 0.7867398858070374, "reward_std": 0.35842904448509216, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22162646055221558, "sampling/sampling_logp_difference/max": 1.291860580444336, "sampling/importance_sampling_ratio/min": 0.30306050181388855, "sampling/importance_sampling_ratio/mean": 1.043230414390564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4850480556488037, "clip_ratio/low_mean": 0.05835868790745735, "clip_ratio/low_min": 0.05835868790745735, "clip_ratio/high_mean": 0.1666986495256424, "clip_ratio/high_max": 0.1666986495256424, "clip_ratio/region_mean": 0.22505733743309975, "reward_total_mean": 0.7867398858070374, "reward_meter_mean": 0.8475043773651123, "reward_meter_std": 0.340963751077652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7867398858070374, "reward_total_composite_std": 0.35842904448509216, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 10.0} {"timestamp_utc": "2026-04-11T19:24:11Z", "mode": "train", "global_step": 11, "epoch": 0.00042477602718566575, "loss": -0.0602, "grad_norm": 18.612279891967773, "learning_rate": 9.96969696969697e-06, "num_tokens": 21060.0, "completions/mean_length": 44.125, "completions/min_length": 30.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.3987988829612732, "rewards/meter/std": 0.39860498905181885, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3987988829612732, "rewards/total_composite/std": 0.39860498905181885, "reward": 0.3987988829612732, "reward_std": 0.39860498905181885, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2347755879163742, "sampling/sampling_logp_difference/max": 1.8758411407470703, "sampling/importance_sampling_ratio/min": 0.15322603285312653, "sampling/importance_sampling_ratio/mean": 1.0170056819915771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.80090893805027, "clip_ratio/low_mean": 0.09850872680544853, "clip_ratio/low_min": 0.09850872680544853, "clip_ratio/high_mean": 0.12178206816315651, "clip_ratio/high_max": 0.12178206816315651, "clip_ratio/region_mean": 0.22029079496860504, "reward_total_mean": 0.3987988829612732, "reward_meter_mean": 0.3987988829612732, "reward_meter_std": 0.39860498905181885, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3987988829612732, "reward_total_composite_std": 0.39860498905181885, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 11.0} {"timestamp_utc": "2026-04-11T19:24:15Z", "mode": "train", "global_step": 12, "epoch": 0.0004633920296570899, "loss": 0.0342, "grad_norm": 18.964420318603516, "learning_rate": 9.966666666666667e-06, "num_tokens": 23077.0, "completions/mean_length": 66.125, "completions/min_length": 55.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.5058028697967529, "rewards/meter/std": 0.40144890546798706, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5058028697967529, "rewards/total_composite/std": 0.40144890546798706, "reward": 0.5058028697967529, "reward_std": 0.40144890546798706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21280689537525177, "sampling/sampling_logp_difference/max": 2.8677234649658203, "sampling/importance_sampling_ratio/min": 0.056828152388334274, "sampling/importance_sampling_ratio/mean": 0.9842817783355713, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9041207674890757, "clip_ratio/low_mean": 0.12340508960187435, "clip_ratio/low_min": 0.12340508960187435, "clip_ratio/high_mean": 0.05123806092888117, "clip_ratio/high_max": 0.05123806092888117, "clip_ratio/region_mean": 0.17464315053075552, "reward_total_mean": 0.5058028697967529, "reward_meter_mean": 0.5058028697967529, "reward_meter_std": 0.40144890546798706, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5058028697967529, "reward_total_composite_std": 0.40144890546798706, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 12.0} {"timestamp_utc": "2026-04-11T19:24:20Z", "mode": "train", "global_step": 13, "epoch": 0.000502008032128514, "loss": 0.2512, "grad_norm": 15.997991561889648, "learning_rate": 9.963636363636364e-06, "num_tokens": 24884.0, "completions/mean_length": 57.875, "completions/min_length": 36.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.45139390230178833, "rewards/meter/std": 0.43168777227401733, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.45139390230178833, "rewards/total_composite/std": 0.43168777227401733, "reward": 0.45139390230178833, "reward_std": 0.4316878020763397, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2293073683977127, "sampling/sampling_logp_difference/max": 1.3626043796539307, "sampling/importance_sampling_ratio/min": 0.25717854499816895, "sampling/importance_sampling_ratio/mean": 1.0242526531219482, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2102586179971695, "clip_ratio/low_mean": 0.08375322818756104, "clip_ratio/low_min": 0.08375322818756104, "clip_ratio/high_mean": 0.1320821400731802, "clip_ratio/high_max": 0.1320821400731802, "clip_ratio/region_mean": 0.21583536826074123, "reward_total_mean": 0.45139390230178833, "reward_meter_mean": 0.45139390230178833, "reward_meter_std": 0.43168777227401733, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.45139390230178833, "reward_total_composite_std": 0.43168777227401733, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 13.0} {"timestamp_utc": "2026-04-11T19:24:25Z", "mode": "train", "global_step": 14, "epoch": 0.0005406240345999382, "loss": -0.029, "grad_norm": 12.638867378234863, "learning_rate": 9.960606060606062e-06, "num_tokens": 27287.0, "completions/mean_length": 107.375, "completions/min_length": 79.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.375, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.5478648543357849, "rewards/meter/std": 0.35016050934791565, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5478648543357849, "rewards/total_composite/std": 0.35016050934791565, "reward": 0.5478648543357849, "reward_std": 0.35016050934791565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2364599108695984, "sampling/sampling_logp_difference/max": 1.8902888298034668, "sampling/importance_sampling_ratio/min": 0.15102817118167877, "sampling/importance_sampling_ratio/mean": 1.019142508506775, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8395927399396896, "clip_ratio/low_mean": 0.09186305850744247, "clip_ratio/low_min": 0.09186305850744247, "clip_ratio/high_mean": 0.10542293824255466, "clip_ratio/high_max": 0.10542293824255466, "clip_ratio/region_mean": 0.19728599674999714, "reward_total_mean": 0.5478648543357849, "reward_meter_mean": 0.5478648543357849, "reward_meter_std": 0.35016050934791565, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5478648543357849, "reward_total_composite_std": 0.35016050934791565, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 14.0} {"timestamp_utc": "2026-04-11T19:24:29Z", "mode": "train", "global_step": 15, "epoch": 0.0005792400370713623, "loss": 0.0038, "grad_norm": 23.416534423828125, "learning_rate": 9.957575757575757e-06, "num_tokens": 28664.0, "completions/mean_length": 25.125, "completions/min_length": 16.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.125, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.4547968804836273, "rewards/meter/std": 0.4638686776161194, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4547968804836273, "rewards/total_composite/std": 0.4638686776161194, "reward": 0.4547968804836273, "reward_std": 0.463868647813797, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17740340530872345, "sampling/sampling_logp_difference/max": 1.0595178604125977, "sampling/importance_sampling_ratio/min": 0.34662288427352905, "sampling/importance_sampling_ratio/mean": 1.0308705568313599, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8547815531492233, "clip_ratio/low_mean": 0.08156565949320793, "clip_ratio/low_min": 0.08156565949320793, "clip_ratio/high_mean": 0.1045012567192316, "clip_ratio/high_max": 0.1045012567192316, "clip_ratio/region_mean": 0.18606691621243954, "reward_total_mean": 0.4547968804836273, "reward_meter_mean": 0.4547968804836273, "reward_meter_std": 0.4638686776161194, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4547968804836273, "reward_total_composite_std": 0.4638686776161194, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 15.0} {"timestamp_utc": "2026-04-11T19:24:33Z", "mode": "train", "global_step": 16, "epoch": 0.0006178560395427865, "loss": 0.1985, "grad_norm": 17.291547775268555, "learning_rate": 9.954545454545456e-06, "num_tokens": 30313.0, "completions/mean_length": 42.125, "completions/min_length": 27.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8227195143699646, "rewards/meter/std": 0.30595460534095764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8227195143699646, "rewards/total_composite/std": 0.30595460534095764, "reward": 0.8227195143699646, "reward_std": 0.30595454573631287, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2161049097776413, "sampling/sampling_logp_difference/max": 1.568324089050293, "sampling/importance_sampling_ratio/min": 0.2083941400051117, "sampling/importance_sampling_ratio/mean": 1.0267019271850586, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8997105956077576, "clip_ratio/low_mean": 0.02777777798473835, "clip_ratio/low_min": 0.02777777798473835, "clip_ratio/high_mean": 0.17772199772298336, "clip_ratio/high_max": 0.17772199772298336, "clip_ratio/region_mean": 0.2054997757077217, "reward_total_mean": 0.8227195143699646, "reward_meter_mean": 0.8227195143699646, "reward_meter_std": 0.30595460534095764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8227195143699646, "reward_total_composite_std": 0.30595460534095764, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 16.0} {"timestamp_utc": "2026-04-11T19:24:37Z", "mode": "train", "global_step": 17, "epoch": 0.0006564720420142106, "loss": -0.0398, "grad_norm": 14.35202693939209, "learning_rate": 9.951515151515152e-06, "num_tokens": 31900.0, "completions/mean_length": 44.375, "completions/min_length": 24.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.375, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8780908584594727, "rewards/meter/std": 0.1825910061597824, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8780908584594727, "rewards/total_composite/std": 0.1825910061597824, "reward": 0.8780908584594727, "reward_std": 0.1825910061597824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09773284941911697, "sampling/sampling_logp_difference/max": 2.0696797370910645, "sampling/importance_sampling_ratio/min": 0.12622620165348053, "sampling/importance_sampling_ratio/mean": 0.9951265454292297, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43756548687815666, "clip_ratio/low_mean": 0.05947580561041832, "clip_ratio/low_min": 0.05947580561041832, "clip_ratio/high_mean": 0.02809617994353175, "clip_ratio/high_max": 0.02809617994353175, "clip_ratio/region_mean": 0.08757198555395007, "reward_total_mean": 0.8780908584594727, "reward_meter_mean": 0.8780908584594727, "reward_meter_std": 0.1825910061597824, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8780908584594727, "reward_total_composite_std": 0.1825910061597824, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 17.0} {"timestamp_utc": "2026-04-11T19:24:42Z", "mode": "train", "global_step": 18, "epoch": 0.0006950880444856349, "loss": 0.2567, "grad_norm": 17.070714950561523, "learning_rate": 9.948484848484849e-06, "num_tokens": 33490.0, "completions/mean_length": 46.75, "completions/min_length": 26.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.45992743968963623, "rewards/meter/std": 0.3919222354888916, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.45992743968963623, "rewards/total_composite/std": 0.3919222354888916, "reward": 0.45992743968963623, "reward_std": 0.3919222354888916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21356253325939178, "sampling/sampling_logp_difference/max": 1.4453849792480469, "sampling/importance_sampling_ratio/min": 0.23565533757209778, "sampling/importance_sampling_ratio/mean": 1.0427944660186768, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.013083055615425, "clip_ratio/low_mean": 0.14138433896005154, "clip_ratio/low_min": 0.14138433896005154, "clip_ratio/high_mean": 0.07820177916437387, "clip_ratio/high_max": 0.07820177916437387, "clip_ratio/region_mean": 0.2195861181244254, "reward_total_mean": 0.45992743968963623, "reward_meter_mean": 0.45992743968963623, "reward_meter_std": 0.3919222354888916, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.45992743968963623, "reward_total_composite_std": 0.3919222354888916, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 18.0} {"timestamp_utc": "2026-04-11T19:24:47Z", "mode": "train", "global_step": 19, "epoch": 0.000733704046957059, "loss": 0.045, "grad_norm": 8.703368186950684, "learning_rate": 9.945454545454546e-06, "num_tokens": 35844.0, "completions/mean_length": 118.25, "completions/min_length": 65.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.25, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.2513314485549927, "rewards/meter/std": 0.1909521371126175, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.22483205795288086, "rewards/total_composite/std": 0.2108626663684845, "reward": 0.22483205795288086, "reward_std": 0.2108626812696457, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20431792736053467, "sampling/sampling_logp_difference/max": 1.916426658630371, "sampling/importance_sampling_ratio/min": 0.1471317857503891, "sampling/importance_sampling_ratio/mean": 1.0284098386764526, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.113861307501793, "clip_ratio/low_mean": 0.1405880395323038, "clip_ratio/low_min": 0.1405880395323038, "clip_ratio/high_mean": 0.05499837175011635, "clip_ratio/high_max": 0.05499837175011635, "clip_ratio/region_mean": 0.19558641128242016, "reward_total_mean": 0.22483205795288086, "reward_meter_mean": 0.2513314485549927, "reward_meter_std": 0.1909521371126175, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.22483205795288086, "reward_total_composite_std": 0.2108626663684845, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 19.0} {"timestamp_utc": "2026-04-11T19:24:51Z", "mode": "train", "global_step": 20, "epoch": 0.0007723200494284832, "loss": 0.0584, "grad_norm": 14.549918174743652, "learning_rate": 9.942424242424244e-06, "num_tokens": 37711.0, "completions/mean_length": 59.375, "completions/min_length": 43.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.5680004358291626, "rewards/meter/std": 0.3963819742202759, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5284043550491333, "rewards/total_composite/std": 0.44208046793937683, "reward": 0.5284043550491333, "reward_std": 0.44208046793937683, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24648143351078033, "sampling/sampling_logp_difference/max": 1.9133062362670898, "sampling/importance_sampling_ratio/min": 0.14759160578250885, "sampling/importance_sampling_ratio/mean": 1.0309125185012817, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4152235835790634, "clip_ratio/low_mean": 0.09608769603073597, "clip_ratio/low_min": 0.09608769603073597, "clip_ratio/high_mean": 0.11636904999613762, "clip_ratio/high_max": 0.11636904999613762, "clip_ratio/region_mean": 0.2124567460268736, "reward_total_mean": 0.5284043550491333, "reward_meter_mean": 0.5680004358291626, "reward_meter_std": 0.3963819742202759, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5284043550491333, "reward_total_composite_std": 0.44208046793937683, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 20.0} {"timestamp_utc": "2026-04-11T19:24:57Z", "mode": "train", "global_step": 21, "epoch": 0.0008109360518999073, "loss": 0.0205, "grad_norm": 6.904236316680908, "learning_rate": 9.939393939393939e-06, "num_tokens": 40580.0, "completions/mean_length": 171.625, "completions/min_length": 126.0, "completions/max_length": 195.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 171.625, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 195.0, "rewards/meter/mean": 0.29830992221832275, "rewards/meter/std": 0.28082314133644104, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.22051820158958435, "rewards/total_composite/std": 0.3221690356731415, "reward": 0.22051820158958435, "reward_std": 0.3221690356731415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18756546080112457, "sampling/sampling_logp_difference/max": 1.4145736694335938, "sampling/importance_sampling_ratio/min": 0.24302920699119568, "sampling/importance_sampling_ratio/mean": 1.0334407091140747, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.130246579647064, "clip_ratio/low_mean": 0.10977440886199474, "clip_ratio/low_min": 0.10977440886199474, "clip_ratio/high_mean": 0.07960096746683121, "clip_ratio/high_max": 0.07960096746683121, "clip_ratio/region_mean": 0.18937537632882595, "reward_total_mean": 0.22051820158958435, "reward_meter_mean": 0.29830992221832275, "reward_meter_std": 0.28082314133644104, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.22051820158958435, "reward_total_composite_std": 0.3221690356731415, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 21.0} {"timestamp_utc": "2026-04-11T19:25:02Z", "mode": "train", "global_step": 22, "epoch": 0.0008495520543713315, "loss": -0.151, "grad_norm": 15.95211124420166, "learning_rate": 9.936363636363638e-06, "num_tokens": 42240.0, "completions/mean_length": 46.5, "completions/min_length": 27.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.48668479919433594, "rewards/meter/std": 0.44610869884490967, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.48668479919433594, "rewards/total_composite/std": 0.44610869884490967, "reward": 0.48668479919433594, "reward_std": 0.4461086690425873, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19472216069698334, "sampling/sampling_logp_difference/max": 1.9752907752990723, "sampling/importance_sampling_ratio/min": 0.13872095942497253, "sampling/importance_sampling_ratio/mean": 1.032152533531189, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.631166860461235, "clip_ratio/low_mean": 0.12670023273676634, "clip_ratio/low_min": 0.12670023273676634, "clip_ratio/high_mean": 0.07555159274488688, "clip_ratio/high_max": 0.07555159274488688, "clip_ratio/region_mean": 0.2022518254816532, "reward_total_mean": 0.48668479919433594, "reward_meter_mean": 0.48668479919433594, "reward_meter_std": 0.44610869884490967, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.48668479919433594, "reward_total_composite_std": 0.44610869884490967, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 22.0} {"timestamp_utc": "2026-04-11T19:25:06Z", "mode": "train", "global_step": 23, "epoch": 0.0008881680568427556, "loss": 0.0144, "grad_norm": 12.759511947631836, "learning_rate": 9.933333333333334e-06, "num_tokens": 44023.0, "completions/mean_length": 59.875, "completions/min_length": 51.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.875, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.5005471706390381, "rewards/meter/std": 0.4535389840602875, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5005471706390381, "rewards/total_composite/std": 0.4535389840602875, "reward": 0.5005471706390381, "reward_std": 0.4535389840602875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20020200312137604, "sampling/sampling_logp_difference/max": 1.3471899032592773, "sampling/importance_sampling_ratio/min": 0.2599697709083557, "sampling/importance_sampling_ratio/mean": 1.0337328910827637, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0983588993549347, "clip_ratio/low_mean": 0.09274027217179537, "clip_ratio/low_min": 0.09274027217179537, "clip_ratio/high_mean": 0.12303969636559486, "clip_ratio/high_max": 0.12303969636559486, "clip_ratio/region_mean": 0.21577996853739023, "reward_total_mean": 0.5005471706390381, "reward_meter_mean": 0.5005471706390381, "reward_meter_std": 0.4535389840602875, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5005471706390381, "reward_total_composite_std": 0.4535389840602875, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 23.0} {"timestamp_utc": "2026-04-11T19:25:11Z", "mode": "train", "global_step": 24, "epoch": 0.0009267840593141798, "loss": 0.0997, "grad_norm": 18.0654354095459, "learning_rate": 9.930303030303031e-06, "num_tokens": 45780.0, "completions/mean_length": 50.625, "completions/min_length": 32.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.6220148801803589, "rewards/meter/std": 0.4630615711212158, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.4580531120300293, "rewards/total_composite/std": 0.46495571732521057, "reward": 0.4580531120300293, "reward_std": 0.46495571732521057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2604425549507141, "sampling/sampling_logp_difference/max": 1.490565299987793, "sampling/importance_sampling_ratio/min": 0.22524529695510864, "sampling/importance_sampling_ratio/mean": 1.0300155878067017, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2059893161058426, "clip_ratio/low_mean": 0.09042087476700544, "clip_ratio/low_min": 0.09042087476700544, "clip_ratio/high_mean": 0.11741364374756813, "clip_ratio/high_max": 0.11741364374756813, "clip_ratio/region_mean": 0.20783451851457357, "reward_total_mean": 0.4580531120300293, "reward_meter_mean": 0.6220148801803589, "reward_meter_std": 0.4630615711212158, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.4580531120300293, "reward_total_composite_std": 0.46495571732521057, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 24.0} {"timestamp_utc": "2026-04-11T19:25:15Z", "mode": "train", "global_step": 25, "epoch": 0.0009654000617856039, "loss": 0.0988, "grad_norm": 15.530243873596191, "learning_rate": 9.927272727272728e-06, "num_tokens": 47479.0, "completions/mean_length": 44.375, "completions/min_length": 28.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.375, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.3527180254459381, "rewards/meter/std": 0.4116312861442566, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3527180254459381, "rewards/total_composite/std": 0.4116312861442566, "reward": 0.3527180254459381, "reward_std": 0.4116312563419342, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17016799747943878, "sampling/sampling_logp_difference/max": 1.5704388618469238, "sampling/importance_sampling_ratio/min": 0.20795388519763947, "sampling/importance_sampling_ratio/mean": 1.016542911529541, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3184484615921974, "clip_ratio/low_mean": 0.09993548225611448, "clip_ratio/low_min": 0.09993548225611448, "clip_ratio/high_mean": 0.08282828330993652, "clip_ratio/high_max": 0.08282828330993652, "clip_ratio/region_mean": 0.182763765566051, "reward_total_mean": 0.3527180254459381, "reward_meter_mean": 0.3527180254459381, "reward_meter_std": 0.4116312861442566, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3527180254459381, "reward_total_composite_std": 0.4116312861442566, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 25.0} {"timestamp_utc": "2026-04-11T19:25:19Z", "mode": "train", "global_step": 26, "epoch": 0.001004016064257028, "loss": -0.082, "grad_norm": 21.890466690063477, "learning_rate": 9.924242424242425e-06, "num_tokens": 49115.0, "completions/mean_length": 44.5, "completions/min_length": 33.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.8464770913124084, "rewards/meter/std": 0.3041004538536072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8331155776977539, "rewards/total_composite/std": 0.3413103222846985, "reward": 0.8331155776977539, "reward_std": 0.3413103222846985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25624310970306396, "sampling/sampling_logp_difference/max": 1.90574312210083, "sampling/importance_sampling_ratio/min": 0.14871208369731903, "sampling/importance_sampling_ratio/mean": 1.0382486581802368, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9691928625106812, "clip_ratio/low_mean": 0.02651515230536461, "clip_ratio/low_min": 0.02651515230536461, "clip_ratio/high_mean": 0.17995928972959518, "clip_ratio/high_max": 0.17995928972959518, "clip_ratio/region_mean": 0.2064744420349598, "reward_total_mean": 0.8331155776977539, "reward_meter_mean": 0.8464770913124084, "reward_meter_std": 0.3041004538536072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8331155776977539, "reward_total_composite_std": 0.3413103222846985, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 26.0} {"timestamp_utc": "2026-04-11T19:25:25Z", "mode": "train", "global_step": 27, "epoch": 0.0010426320667284523, "loss": 0.0432, "grad_norm": 8.763998031616211, "learning_rate": 9.921212121212121e-06, "num_tokens": 51914.0, "completions/mean_length": 159.875, "completions/min_length": 145.0, "completions/max_length": 176.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.875, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 176.0, "rewards/meter/mean": 0.44596123695373535, "rewards/meter/std": 0.3097141981124878, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.44596123695373535, "rewards/total_composite/std": 0.3097141981124878, "reward": 0.44596123695373535, "reward_std": 0.3097141981124878, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19997085630893707, "sampling/sampling_logp_difference/max": 1.782116413116455, "sampling/importance_sampling_ratio/min": 0.16828162968158722, "sampling/importance_sampling_ratio/mean": 1.0340567827224731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.940888598561287, "clip_ratio/low_mean": 0.04641487076878548, "clip_ratio/low_min": 0.04641487076878548, "clip_ratio/high_mean": 0.15723814070224762, "clip_ratio/high_max": 0.15723814070224762, "clip_ratio/region_mean": 0.2036530114710331, "reward_total_mean": 0.44596123695373535, "reward_meter_mean": 0.44596123695373535, "reward_meter_std": 0.3097141981124878, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.44596123695373535, "reward_total_composite_std": 0.3097141981124878, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 27.0} {"timestamp_utc": "2026-04-11T19:25:31Z", "mode": "train", "global_step": 28, "epoch": 0.0010812480691998764, "loss": -0.0001, "grad_norm": 16.314180374145508, "learning_rate": 9.918181818181818e-06, "num_tokens": 54507.0, "completions/mean_length": 122.125, "completions/min_length": 89.0, "completions/max_length": 223.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.125, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 223.0, "rewards/meter/mean": 0.4260466396808624, "rewards/meter/std": 0.2623087763786316, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0578637570142746, "rewards/arabic_clean/mean": 0.375, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.06344515830278397, "rewards/total_composite/std": 0.09849678725004196, "reward": 0.06344515830278397, "reward_std": 0.09849678725004196, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2858719229698181, "sampling/sampling_logp_difference/max": 3.241093635559082, "sampling/importance_sampling_ratio/min": 0.03912108764052391, "sampling/importance_sampling_ratio/mean": 0.9840916395187378, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0821578055620193, "clip_ratio/low_mean": 0.10222199466079473, "clip_ratio/low_min": 0.10222199466079473, "clip_ratio/high_mean": 0.08741745911538601, "clip_ratio/high_max": 0.08741745911538601, "clip_ratio/region_mean": 0.18963945377618074, "reward_total_mean": 0.06344515830278397, "reward_meter_mean": 0.4260466396808624, "reward_meter_std": 0.2623087763786316, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0578637570142746, "reward_arabic_clean_mean": 0.375, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.06344515830278397, "reward_total_composite_std": 0.09849678725004196, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 28.0} {"timestamp_utc": "2026-04-11T19:25:35Z", "mode": "train", "global_step": 29, "epoch": 0.0011198640716713006, "loss": 0.0883, "grad_norm": 18.285869598388672, "learning_rate": 9.915151515151515e-06, "num_tokens": 56313.0, "completions/mean_length": 44.75, "completions/min_length": 33.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.7853162884712219, "rewards/meter/std": 0.29714658856391907, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7853162884712219, "rewards/total_composite/std": 0.29714658856391907, "reward": 0.7853162884712219, "reward_std": 0.29714658856391907, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20776626467704773, "sampling/sampling_logp_difference/max": 2.2629013061523438, "sampling/importance_sampling_ratio/min": 0.10404817014932632, "sampling/importance_sampling_ratio/mean": 0.9876272082328796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6001797169446945, "clip_ratio/low_mean": 0.05003259517252445, "clip_ratio/low_min": 0.05003259517252445, "clip_ratio/high_mean": 0.17736977525055408, "clip_ratio/high_max": 0.17736977525055408, "clip_ratio/region_mean": 0.22740237042307854, "reward_total_mean": 0.7853162884712219, "reward_meter_mean": 0.7853162884712219, "reward_meter_std": 0.29714658856391907, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7853162884712219, "reward_total_composite_std": 0.29714658856391907, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 29.0} {"timestamp_utc": "2026-04-11T19:25:42Z", "mode": "train", "global_step": 30, "epoch": 0.0011584800741427247, "loss": 0.0412, "grad_norm": 8.281560897827148, "learning_rate": 9.912121212121213e-06, "num_tokens": 59278.0, "completions/mean_length": 169.625, "completions/min_length": 127.0, "completions/max_length": 220.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 169.625, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 220.0, "rewards/meter/mean": 0.5359864234924316, "rewards/meter/std": 0.21292957663536072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.48327285051345825, "rewards/total_composite/std": 0.28519800305366516, "reward": 0.48327285051345825, "reward_std": 0.2851979732513428, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.227935329079628, "sampling/sampling_logp_difference/max": 1.7696037292480469, "sampling/importance_sampling_ratio/min": 0.1704005002975464, "sampling/importance_sampling_ratio/mean": 1.0370231866836548, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6212728768587112, "clip_ratio/low_mean": 0.09173617325723171, "clip_ratio/low_min": 0.09173617325723171, "clip_ratio/high_mean": 0.1195354238152504, "clip_ratio/high_max": 0.1195354238152504, "clip_ratio/region_mean": 0.2112715970724821, "reward_total_mean": 0.48327285051345825, "reward_meter_mean": 0.5359864234924316, "reward_meter_std": 0.21292957663536072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.48327285051345825, "reward_total_composite_std": 0.28519800305366516, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 30.0} {"timestamp_utc": "2026-04-11T19:25:46Z", "mode": "train", "global_step": 31, "epoch": 0.001197096076614149, "loss": 0.0224, "grad_norm": 14.505411148071289, "learning_rate": 9.90909090909091e-06, "num_tokens": 61117.0, "completions/mean_length": 67.875, "completions/min_length": 53.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.6783833503723145, "rewards/meter/std": 0.4296880066394806, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6783833503723145, "rewards/total_composite/std": 0.4296880066394806, "reward": 0.6783833503723145, "reward_std": 0.4296879768371582, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24672164022922516, "sampling/sampling_logp_difference/max": 1.5443954467773438, "sampling/importance_sampling_ratio/min": 0.2134408801794052, "sampling/importance_sampling_ratio/mean": 1.0112130641937256, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.164387509226799, "clip_ratio/low_mean": 0.06649214401841164, "clip_ratio/low_min": 0.06649214401841164, "clip_ratio/high_mean": 0.16296951659023762, "clip_ratio/high_max": 0.16296951659023762, "clip_ratio/region_mean": 0.22946166060864925, "reward_total_mean": 0.6783833503723145, "reward_meter_mean": 0.6783833503723145, "reward_meter_std": 0.4296880066394806, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6783833503723145, "reward_total_composite_std": 0.4296880066394806, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 31.0} {"timestamp_utc": "2026-04-11T19:25:51Z", "mode": "train", "global_step": 32, "epoch": 0.001235712079085573, "loss": 0.0377, "grad_norm": 22.633512496948242, "learning_rate": 9.906060606060607e-06, "num_tokens": 62698.0, "completions/mean_length": 27.625, "completions/min_length": 25.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9473069310188293, "rewards/meter/std": 0.03623950853943825, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9473069310188293, "rewards/total_composite/std": 0.03623950853943825, "reward": 0.9473069310188293, "reward_std": 0.03623950108885765, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09311103075742722, "sampling/sampling_logp_difference/max": 2.3337249755859375, "sampling/importance_sampling_ratio/min": 0.0969339981675148, "sampling/importance_sampling_ratio/mean": 0.9921619892120361, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44207035191357136, "clip_ratio/low_mean": 0.03902714978903532, "clip_ratio/low_min": 0.03902714978903532, "clip_ratio/high_mean": 0.02909391513094306, "clip_ratio/high_max": 0.02909391513094306, "clip_ratio/region_mean": 0.06812106491997838, "reward_total_mean": 0.9473069310188293, "reward_meter_mean": 0.9473069310188293, "reward_meter_std": 0.03623950853943825, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9473069310188293, "reward_total_composite_std": 0.03623950853943825, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 32.0} {"timestamp_utc": "2026-04-11T19:25:58Z", "mode": "train", "global_step": 33, "epoch": 0.0012743280815569972, "loss": -0.1309, "grad_norm": 9.995526313781738, "learning_rate": 9.903030303030305e-06, "num_tokens": 65811.0, "completions/mean_length": 181.125, "completions/min_length": 134.0, "completions/max_length": 274.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.125, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 274.0, "rewards/meter/mean": 0.2281605750322342, "rewards/meter/std": 0.13796716928482056, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.05750546231865883, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.19121383130550385, "rewards/total_composite/std": 0.15800805389881134, "reward": 0.19121383130550385, "reward_std": 0.15800805389881134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.26636290550231934, "sampling/sampling_logp_difference/max": 2.2858448028564453, "sampling/importance_sampling_ratio/min": 0.10168811678886414, "sampling/importance_sampling_ratio/mean": 1.0408287048339844, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.361170306801796, "clip_ratio/low_mean": 0.14664593152701855, "clip_ratio/low_min": 0.14664593152701855, "clip_ratio/high_mean": 0.08508810587227345, "clip_ratio/high_max": 0.08508810587227345, "clip_ratio/region_mean": 0.231734037399292, "reward_total_mean": 0.19121383130550385, "reward_meter_mean": 0.2281605750322342, "reward_meter_std": 0.13796716928482056, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.05750546231865883, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.19121383130550385, "reward_total_composite_std": 0.15800805389881134, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 33.0} {"timestamp_utc": "2026-04-11T19:26:02Z", "mode": "train", "global_step": 34, "epoch": 0.0013129440840284213, "loss": 0.0783, "grad_norm": 21.821182250976562, "learning_rate": 9.9e-06, "num_tokens": 67182.0, "completions/mean_length": 26.375, "completions/min_length": 23.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.1796203851699829, "rewards/meter/std": 0.330172061920166, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.1796203851699829, "rewards/total_composite/std": 0.330172061920166, "reward": 0.1796203851699829, "reward_std": 0.330172061920166, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18016308546066284, "sampling/sampling_logp_difference/max": 2.0957188606262207, "sampling/importance_sampling_ratio/min": 0.12298180162906647, "sampling/importance_sampling_ratio/mean": 1.0188137292861938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5113434419035912, "clip_ratio/low_mean": 0.09407370304688811, "clip_ratio/low_min": 0.09407370304688811, "clip_ratio/high_mean": 0.03525641094893217, "clip_ratio/high_max": 0.03525641094893217, "clip_ratio/region_mean": 0.12933011399582028, "reward_total_mean": 0.1796203851699829, "reward_meter_mean": 0.1796203851699829, "reward_meter_std": 0.330172061920166, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.1796203851699829, "reward_total_composite_std": 0.330172061920166, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 34.0} {"timestamp_utc": "2026-04-11T19:26:06Z", "mode": "train", "global_step": 35, "epoch": 0.0013515600864998456, "loss": -0.0922, "grad_norm": 14.014320373535156, "learning_rate": 9.896969696969699e-06, "num_tokens": 68837.0, "completions/mean_length": 45.875, "completions/min_length": 33.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.5994056463241577, "rewards/meter/std": 0.40437188744544983, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5994056463241577, "rewards/total_composite/std": 0.40437188744544983, "reward": 0.5994056463241577, "reward_std": 0.40437188744544983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1685187667608261, "sampling/sampling_logp_difference/max": 1.3427734375, "sampling/importance_sampling_ratio/min": 0.31121668219566345, "sampling/importance_sampling_ratio/mean": 1.0345124006271362, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3972373455762863, "clip_ratio/low_mean": 0.05922202859073877, "clip_ratio/low_min": 0.05922202859073877, "clip_ratio/high_mean": 0.07765224599279463, "clip_ratio/high_max": 0.07765224599279463, "clip_ratio/region_mean": 0.1368742745835334, "reward_total_mean": 0.5994056463241577, "reward_meter_mean": 0.5994056463241577, "reward_meter_std": 0.40437188744544983, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5994056463241577, "reward_total_composite_std": 0.40437188744544983, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 35.0} {"timestamp_utc": "2026-04-11T19:26:17Z", "mode": "train", "global_step": 36, "epoch": 0.0013901760889712697, "loss": 0.2225, "grad_norm": 3.0867953300476074, "learning_rate": 9.893939393939395e-06, "num_tokens": 70919.0, "completions/mean_length": 497.25, "completions/min_length": 394.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.875, "completions/mean_terminated_length": 394.0, "completions/min_terminated_length": 394.0, "completions/max_terminated_length": 394.0, "rewards/meter/mean": 0.86127769947052, "rewards/meter/std": 0.20221967995166779, "rewards/count_adherence/mean": 0.9264706373214722, "rewards/count_adherence/std": 0.08753220736980438, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6741494536399841, "rewards/total_composite/std": 0.3219583332538605, "reward": 0.6741494536399841, "reward_std": 0.3219583332538605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2478996217250824, "sampling/sampling_logp_difference/max": 1.2088394165039062, "sampling/importance_sampling_ratio/min": 0.2985435426235199, "sampling/importance_sampling_ratio/mean": 1.0490732192993164, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48652103543281555, "clip_ratio/low_mean": 0.02442893385887146, "clip_ratio/low_min": 0.02442893385887146, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.02442893385887146, "reward_total_mean": 0.6741494536399841, "reward_meter_mean": 0.86127769947052, "reward_meter_std": 0.20221967995166779, "reward_count_adherence_mean": 0.9264706373214722, "reward_count_adherence_std": 0.08753220736980438, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6741494536399841, "reward_total_composite_std": 0.3219583332538605, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 36.0} {"timestamp_utc": "2026-04-11T19:26:21Z", "mode": "train", "global_step": 37, "epoch": 0.0014287920914426938, "loss": 0.0523, "grad_norm": 15.039297103881836, "learning_rate": 9.890909090909092e-06, "num_tokens": 72813.0, "completions/mean_length": 66.75, "completions/min_length": 50.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.75, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.6241669654846191, "rewards/meter/std": 0.3833092451095581, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5642927289009094, "rewards/total_composite/std": 0.3661707043647766, "reward": 0.5642927289009094, "reward_std": 0.3661707043647766, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2267686128616333, "sampling/sampling_logp_difference/max": 1.5563976764678955, "sampling/importance_sampling_ratio/min": 0.2108944207429886, "sampling/importance_sampling_ratio/mean": 1.026777744293213, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.081456035375595, "clip_ratio/low_mean": 0.09653645940124989, "clip_ratio/low_min": 0.09653645940124989, "clip_ratio/high_mean": 0.1309125702828169, "clip_ratio/high_max": 0.1309125702828169, "clip_ratio/region_mean": 0.22744902968406677, "reward_total_mean": 0.5642927289009094, "reward_meter_mean": 0.6241669654846191, "reward_meter_std": 0.3833092451095581, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5642927289009094, "reward_total_composite_std": 0.3661707043647766, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 37.0} {"timestamp_utc": "2026-04-11T19:26:25Z", "mode": "train", "global_step": 38, "epoch": 0.001467408093914118, "loss": 0.0441, "grad_norm": 17.81291389465332, "learning_rate": 9.887878787878789e-06, "num_tokens": 74327.0, "completions/mean_length": 26.25, "completions/min_length": 23.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.25, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9193331003189087, "rewards/meter/std": 0.18450921773910522, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9193331003189087, "rewards/total_composite/std": 0.18450921773910522, "reward": 0.9193331003189087, "reward_std": 0.18450923264026642, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18091736733913422, "sampling/sampling_logp_difference/max": 2.5841870307922363, "sampling/importance_sampling_ratio/min": 0.07545739412307739, "sampling/importance_sampling_ratio/mean": 0.9957528710365295, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.365403652191162, "clip_ratio/low_mean": 0.0223214291036129, "clip_ratio/low_min": 0.0223214291036129, "clip_ratio/high_mean": 0.13196256244555116, "clip_ratio/high_max": 0.13196256244555116, "clip_ratio/region_mean": 0.15428399154916406, "reward_total_mean": 0.9193331003189087, "reward_meter_mean": 0.9193331003189087, "reward_meter_std": 0.18450921773910522, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9193331003189087, "reward_total_composite_std": 0.18450921773910522, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 38.0} {"timestamp_utc": "2026-04-11T19:26:30Z", "mode": "train", "global_step": 39, "epoch": 0.0015060240963855422, "loss": 0.0578, "grad_norm": 11.80769157409668, "learning_rate": 9.884848484848486e-06, "num_tokens": 76302.0, "completions/mean_length": 82.875, "completions/min_length": 76.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.875, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.2643655240535736, "rewards/meter/std": 0.3247142732143402, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.2587703466415405, "rewards/total_composite/std": 0.32939082384109497, "reward": 0.2587703466415405, "reward_std": 0.3293907940387726, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20199847221374512, "sampling/sampling_logp_difference/max": 11.316105842590332, "sampling/importance_sampling_ratio/min": 1.2175244592071977e-05, "sampling/importance_sampling_ratio/mean": 0.9873656630516052, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.739346232265234, "clip_ratio/low_mean": 0.10097905434668064, "clip_ratio/low_min": 0.10097905434668064, "clip_ratio/high_mean": 0.03838060609996319, "clip_ratio/high_max": 0.03838060609996319, "clip_ratio/region_mean": 0.13935966044664383, "reward_total_mean": 0.2587703466415405, "reward_meter_mean": 0.2643655240535736, "reward_meter_std": 0.3247142732143402, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.2587703466415405, "reward_total_composite_std": 0.32939082384109497, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 39.0} {"timestamp_utc": "2026-04-11T19:26:35Z", "mode": "train", "global_step": 40, "epoch": 0.0015446400988569664, "loss": -0.0041, "grad_norm": 9.70837116241455, "learning_rate": 9.881818181818182e-06, "num_tokens": 78586.0, "completions/mean_length": 107.5, "completions/min_length": 77.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.5, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.8742237687110901, "rewards/meter/std": 0.31200793385505676, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8742237687110901, "rewards/total_composite/std": 0.31200793385505676, "reward": 0.8742237687110901, "reward_std": 0.31200793385505676, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21297502517700195, "sampling/sampling_logp_difference/max": 1.6355867385864258, "sampling/importance_sampling_ratio/min": 0.1948380321264267, "sampling/importance_sampling_ratio/mean": 1.039535641670227, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.759167045354843, "clip_ratio/low_mean": 0.031887754797935486, "clip_ratio/low_min": 0.031887754797935486, "clip_ratio/high_mean": 0.18653450906276703, "clip_ratio/high_max": 0.18653450906276703, "clip_ratio/region_mean": 0.21842226386070251, "reward_total_mean": 0.8742237687110901, "reward_meter_mean": 0.8742237687110901, "reward_meter_std": 0.31200793385505676, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8742237687110901, "reward_total_composite_std": 0.31200793385505676, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 40.0} {"timestamp_utc": "2026-04-11T19:26:44Z", "mode": "train", "global_step": 41, "epoch": 0.0015832561013283905, "loss": 0.0665, "grad_norm": 5.116574287414551, "learning_rate": 9.87878787878788e-06, "num_tokens": 83433.0, "completions/mean_length": 390.875, "completions/min_length": 322.0, "completions/max_length": 433.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 390.875, "completions/min_terminated_length": 322.0, "completions/max_terminated_length": 433.0, "rewards/meter/mean": 0.7916162014007568, "rewards/meter/std": 0.1887982189655304, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.07559289038181305, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6314187049865723, "rewards/total_composite/std": 0.3165915012359619, "reward": 0.6314187049865723, "reward_std": 0.3165914714336395, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19484373927116394, "sampling/sampling_logp_difference/max": 1.7193355560302734, "sampling/importance_sampling_ratio/min": 0.1791851669549942, "sampling/importance_sampling_ratio/mean": 1.0329089164733887, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.349475011229515, "clip_ratio/low_mean": 0.08622571267187595, "clip_ratio/low_min": 0.08622571267187595, "clip_ratio/high_mean": 0.09009376727044582, "clip_ratio/high_max": 0.09009376727044582, "clip_ratio/region_mean": 0.17631947994232178, "reward_total_mean": 0.6314187049865723, "reward_meter_mean": 0.7916162014007568, "reward_meter_std": 0.1887982189655304, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.07559289038181305, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6314187049865723, "reward_total_composite_std": 0.3165915012359619, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 41.0} {"timestamp_utc": "2026-04-11T19:26:49Z", "mode": "train", "global_step": 42, "epoch": 0.0016218721037998146, "loss": 0.1992, "grad_norm": 13.098201751708984, "learning_rate": 9.875757575757576e-06, "num_tokens": 85365.0, "completions/mean_length": 71.5, "completions/min_length": 48.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.5, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.384579598903656, "rewards/meter/std": 0.38699981570243835, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.384579598903656, "rewards/total_composite/std": 0.38699981570243835, "reward": 0.384579598903656, "reward_std": 0.38699978590011597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.216371089220047, "sampling/sampling_logp_difference/max": 2.177125930786133, "sampling/importance_sampling_ratio/min": 0.11336687952280045, "sampling/importance_sampling_ratio/mean": 1.0288784503936768, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.382046639919281, "clip_ratio/low_mean": 0.12379511073231697, "clip_ratio/low_min": 0.12379511073231697, "clip_ratio/high_mean": 0.09933997690677643, "clip_ratio/high_max": 0.09933997690677643, "clip_ratio/region_mean": 0.2231350876390934, "reward_total_mean": 0.384579598903656, "reward_meter_mean": 0.384579598903656, "reward_meter_std": 0.38699981570243835, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.384579598903656, "reward_total_composite_std": 0.38699981570243835, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 42.0} {"timestamp_utc": "2026-04-11T19:26:53Z", "mode": "train", "global_step": 43, "epoch": 0.0016604881062712389, "loss": -0.0628, "grad_norm": 19.66609764099121, "learning_rate": 9.872727272727274e-06, "num_tokens": 87028.0, "completions/mean_length": 52.875, "completions/min_length": 25.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.875, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.40058913826942444, "rewards/meter/std": 0.4409148097038269, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.40058913826942444, "rewards/total_composite/std": 0.4409148097038269, "reward": 0.40058913826942444, "reward_std": 0.4409147799015045, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22723107039928436, "sampling/sampling_logp_difference/max": 1.5011954307556152, "sampling/importance_sampling_ratio/min": 0.2228635996580124, "sampling/importance_sampling_ratio/mean": 1.0391393899917603, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.689612329006195, "clip_ratio/low_mean": 0.14910552836954594, "clip_ratio/low_min": 0.14910552836954594, "clip_ratio/high_mean": 0.07029963098466396, "clip_ratio/high_max": 0.07029963098466396, "clip_ratio/region_mean": 0.2194051593542099, "reward_total_mean": 0.40058913826942444, "reward_meter_mean": 0.40058913826942444, "reward_meter_std": 0.4409148097038269, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.40058913826942444, "reward_total_composite_std": 0.4409148097038269, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 43.0} {"timestamp_utc": "2026-04-11T19:26:58Z", "mode": "train", "global_step": 44, "epoch": 0.001699104108742663, "loss": 0.0541, "grad_norm": 20.52503204345703, "learning_rate": 9.869696969696971e-06, "num_tokens": 88767.0, "completions/mean_length": 45.375, "completions/min_length": 43.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6814202070236206, "rewards/meter/std": 0.4100963771343231, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6814202070236206, "rewards/total_composite/std": 0.4100963771343231, "reward": 0.6814202070236206, "reward_std": 0.4100963771343231, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10806272178888321, "sampling/sampling_logp_difference/max": 1.696730136871338, "sampling/importance_sampling_ratio/min": 0.18328185379505157, "sampling/importance_sampling_ratio/mean": 0.9983127117156982, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40569959953427315, "clip_ratio/low_mean": 0.0339520201086998, "clip_ratio/low_min": 0.0339520201086998, "clip_ratio/high_mean": 0.041567519307136536, "clip_ratio/high_max": 0.041567519307136536, "clip_ratio/region_mean": 0.07551953941583633, "reward_total_mean": 0.6814202070236206, "reward_meter_mean": 0.6814202070236206, "reward_meter_std": 0.4100963771343231, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6814202070236206, "reward_total_composite_std": 0.4100963771343231, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 44.0} {"timestamp_utc": "2026-04-11T19:27:02Z", "mode": "train", "global_step": 45, "epoch": 0.001737720111214087, "loss": 0.2204, "grad_norm": 19.593364715576172, "learning_rate": 9.866666666666668e-06, "num_tokens": 90628.0, "completions/mean_length": 48.625, "completions/min_length": 32.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6895775198936462, "rewards/meter/std": 0.43211013078689575, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6895775198936462, "rewards/total_composite/std": 0.43211013078689575, "reward": 0.6895775198936462, "reward_std": 0.43211016058921814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22784224152565002, "sampling/sampling_logp_difference/max": 1.557765007019043, "sampling/importance_sampling_ratio/min": 0.21060624718666077, "sampling/importance_sampling_ratio/mean": 1.0464413166046143, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1141539961099625, "clip_ratio/low_mean": 0.07070140354335308, "clip_ratio/low_min": 0.07070140354335308, "clip_ratio/high_mean": 0.12581374496221542, "clip_ratio/high_max": 0.12581374496221542, "clip_ratio/region_mean": 0.1965151485055685, "reward_total_mean": 0.6895775198936462, "reward_meter_mean": 0.6895775198936462, "reward_meter_std": 0.43211013078689575, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6895775198936462, "reward_total_composite_std": 0.43211013078689575, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 45.0} {"timestamp_utc": "2026-04-11T19:27:06Z", "mode": "train", "global_step": 46, "epoch": 0.0017763361136855112, "loss": 0.0451, "grad_norm": 24.487468719482422, "learning_rate": 9.863636363636364e-06, "num_tokens": 92020.0, "completions/mean_length": 36.0, "completions/min_length": 27.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.22091448307037354, "rewards/meter/std": 0.17998214066028595, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.2181696593761444, "rewards/total_composite/std": 0.18358123302459717, "reward": 0.2181696593761444, "reward_std": 0.18358123302459717, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22377783060073853, "sampling/sampling_logp_difference/max": 1.75813627243042, "sampling/importance_sampling_ratio/min": 0.17236579954624176, "sampling/importance_sampling_ratio/mean": 0.9994174242019653, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3433509916067123, "clip_ratio/low_mean": 0.11509755253791809, "clip_ratio/low_min": 0.11509755253791809, "clip_ratio/high_mean": 0.09895163122564554, "clip_ratio/high_max": 0.09895163122564554, "clip_ratio/region_mean": 0.21404918376356363, "reward_total_mean": 0.2181696593761444, "reward_meter_mean": 0.22091448307037354, "reward_meter_std": 0.17998214066028595, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.2181696593761444, "reward_total_composite_std": 0.18358123302459717, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 46.0} {"timestamp_utc": "2026-04-11T19:27:10Z", "mode": "train", "global_step": 47, "epoch": 0.0018149521161569355, "loss": -0.0284, "grad_norm": 19.872201919555664, "learning_rate": 9.860606060606061e-06, "num_tokens": 93400.0, "completions/mean_length": 25.5, "completions/min_length": 24.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.5, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.8368544578552246, "rewards/meter/std": 0.035664137452840805, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8368544578552246, "rewards/total_composite/std": 0.035664137452840805, "reward": 0.8368544578552246, "reward_std": 0.0356641449034214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07806604355573654, "sampling/sampling_logp_difference/max": 3.2258620262145996, "sampling/importance_sampling_ratio/min": 0.0397215262055397, "sampling/importance_sampling_ratio/mean": 0.9922424554824829, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1728968620300293, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0347222238779068, "clip_ratio/high_max": 0.0347222238779068, "clip_ratio/region_mean": 0.0347222238779068, "reward_total_mean": 0.8368544578552246, "reward_meter_mean": 0.8368544578552246, "reward_meter_std": 0.035664137452840805, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8368544578552246, "reward_total_composite_std": 0.035664137452840805, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 47.0} {"timestamp_utc": "2026-04-11T19:27:16Z", "mode": "train", "global_step": 48, "epoch": 0.0018535681186283596, "loss": 0.0975, "grad_norm": 13.035599708557129, "learning_rate": 9.857575757575758e-06, "num_tokens": 95529.0, "completions/mean_length": 95.125, "completions/min_length": 63.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.125, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.3589945435523987, "rewards/meter/std": 0.3197920322418213, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3589945435523987, "rewards/total_composite/std": 0.3197920322418213, "reward": 0.3589945435523987, "reward_std": 0.3197920024394989, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23110564053058624, "sampling/sampling_logp_difference/max": 2.098705291748047, "sampling/importance_sampling_ratio/min": 0.12261507660150528, "sampling/importance_sampling_ratio/mean": 1.0021162033081055, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.133227914571762, "clip_ratio/low_mean": 0.09916602075099945, "clip_ratio/low_min": 0.09916602075099945, "clip_ratio/high_mean": 0.13037441484630108, "clip_ratio/high_max": 0.13037441484630108, "clip_ratio/region_mean": 0.22954043559730053, "reward_total_mean": 0.3589945435523987, "reward_meter_mean": 0.3589945435523987, "reward_meter_std": 0.3197920322418213, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3589945435523987, "reward_total_composite_std": 0.3197920322418213, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 48.0} {"timestamp_utc": "2026-04-11T19:27:20Z", "mode": "train", "global_step": 49, "epoch": 0.0018921841210997837, "loss": 0.0542, "grad_norm": 15.745153427124023, "learning_rate": 9.854545454545456e-06, "num_tokens": 97325.0, "completions/mean_length": 48.5, "completions/min_length": 31.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.5693104267120361, "rewards/meter/std": 0.4606866240501404, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5693104267120361, "rewards/total_composite/std": 0.4606866240501404, "reward": 0.5693104267120361, "reward_std": 0.460686594247818, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2219070941209793, "sampling/sampling_logp_difference/max": 1.8588149547576904, "sampling/importance_sampling_ratio/min": 0.15585720539093018, "sampling/importance_sampling_ratio/mean": 1.031612515449524, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.332765683531761, "clip_ratio/low_mean": 0.0817372314631939, "clip_ratio/low_min": 0.0817372314631939, "clip_ratio/high_mean": 0.13691459875553846, "clip_ratio/high_max": 0.13691459875553846, "clip_ratio/region_mean": 0.21865183021873236, "reward_total_mean": 0.5693104267120361, "reward_meter_mean": 0.5693104267120361, "reward_meter_std": 0.4606866240501404, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5693104267120361, "reward_total_composite_std": 0.4606866240501404, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 49.0} {"timestamp_utc": "2026-04-11T19:27:26Z", "mode": "train", "global_step": 50, "epoch": 0.0019308001235712078, "loss": -0.0192, "grad_norm": 7.936125755310059, "learning_rate": 9.851515151515151e-06, "num_tokens": 100154.0, "completions/mean_length": 176.625, "completions/min_length": 93.0, "completions/max_length": 214.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 176.625, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 214.0, "rewards/meter/mean": 0.7158485651016235, "rewards/meter/std": 0.3752211630344391, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5725899934768677, "rewards/total_composite/std": 0.41487547755241394, "reward": 0.5725899934768677, "reward_std": 0.41487547755241394, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19521863758563995, "sampling/sampling_logp_difference/max": 2.292977809906006, "sampling/importance_sampling_ratio/min": 0.10096535831689835, "sampling/importance_sampling_ratio/mean": 1.021142601966858, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.019968628883362, "clip_ratio/low_mean": 0.06395815405994654, "clip_ratio/low_min": 0.06395815405994654, "clip_ratio/high_mean": 0.13284815661609173, "clip_ratio/high_max": 0.13284815661609173, "clip_ratio/region_mean": 0.19680631067603827, "reward_total_mean": 0.5725899934768677, "reward_meter_mean": 0.7158485651016235, "reward_meter_std": 0.3752211630344391, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5725899934768677, "reward_total_composite_std": 0.41487547755241394, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 50.0} {"timestamp_utc": "2026-04-11T19:28:45Z", "mode": "eval", "global_step": 50, "epoch": 0.0019308001235712078, "eval_loss": NaN, "eval_runtime": 79.1676, "eval_samples_per_second": 1.314, "eval_steps_per_second": 0.164, "eval_num_tokens": 100154.0, "eval_completions/mean_length": 184.7403846153846, "eval_completions/min_length": 49.30769230769231, "eval_completions/max_length": 426.38461538461536, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/mean_terminated_length": 164.79945608285757, "eval_completions/min_terminated_length": 49.30769230769231, "eval_completions/max_terminated_length": 346.46153846153845, "eval_rewards/meter/mean": 0.4141139342234685, "eval_rewards/meter/std": 0.3737386694321266, "eval_rewards/count_adherence/mean": 0.9684277910452622, "eval_rewards/count_adherence/std": 0.059289679647638246, "eval_rewards/arabic_clean/mean": 0.9230769230769231, "eval_rewards/arabic_clean/std": 0.1987869510283837, "eval_rewards/total_composite/mean": 0.3838638663291931, "eval_rewards/total_composite/std": 0.3831195647899921, "eval_reward": 0.3838638663291931, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.14078256086661264, "eval_sampling/sampling_logp_difference/max": 1.170855081998385, "eval_sampling/importance_sampling_ratio/min": 0.31346100110274094, "eval_sampling/importance_sampling_ratio/mean": 1.0421396860709558, "eval_sampling/importance_sampling_ratio/max": 1.6281351492955134, "eval_entropy": 2.3565903168458204, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.3838638663291931, "eval_reward_meter_mean": 0.4141139342234685, "eval_reward_meter_std": 0.3737386694321266, "eval_reward_count_adherence_mean": 0.9684277910452622, "eval_reward_count_adherence_std": 0.059289679647638246, "eval_reward_arabic_clean_mean": 0.9230769230769231, "eval_reward_arabic_clean_std": 0.1987869510283837, "eval_reward_total_composite_mean": 0.3838638663291931, "eval_reward_total_composite_std": 0.3831195647899921, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 50.0} {"timestamp_utc": "2026-04-11T19:28:52Z", "mode": "train", "global_step": 51, "epoch": 0.001969416126042632, "loss": 0.0165, "grad_norm": 18.508319854736328, "learning_rate": 9.84848484848485e-06, "num_tokens": 101753.0, "completions/mean_length": 41.875, "completions/min_length": 31.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.23897752165794373, "rewards/meter/std": 0.3632267415523529, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.23897752165794373, "rewards/total_composite/std": 0.3632267415523529, "reward": 0.23897752165794373, "reward_std": 0.3632267415523529, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2588263154029846, "sampling/sampling_logp_difference/max": 2.430152177810669, "sampling/importance_sampling_ratio/min": 0.08802343904972076, "sampling/importance_sampling_ratio/mean": 1.0173118114471436, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8089727088809013, "clip_ratio/low_mean": 0.16671104542911053, "clip_ratio/low_min": 0.16671104542911053, "clip_ratio/high_mean": 0.06337687559425831, "clip_ratio/high_max": 0.06337687559425831, "clip_ratio/region_mean": 0.23008792102336884, "reward_total_mean": 0.23897752165794373, "reward_meter_mean": 0.23897752165794373, "reward_meter_std": 0.3632267415523529, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.23897752165794373, "reward_total_composite_std": 0.3632267415523529, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 51.0} {"timestamp_utc": "2026-04-11T19:28:57Z", "mode": "train", "global_step": 52, "epoch": 0.002008032128514056, "loss": -0.048, "grad_norm": 11.88036060333252, "learning_rate": 9.845454545454546e-06, "num_tokens": 103684.0, "completions/mean_length": 67.375, "completions/min_length": 57.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.375, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9933522939682007, "rewards/meter/std": 0.004396223928779364, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9933522939682007, "rewards/total_composite/std": 0.004396223928779364, "reward": 0.9933522939682007, "reward_std": 0.004396222531795502, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20311212539672852, "sampling/sampling_logp_difference/max": 1.5028152465820312, "sampling/importance_sampling_ratio/min": 0.22250288724899292, "sampling/importance_sampling_ratio/mean": 1.0523029565811157, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2769337445497513, "clip_ratio/low_mean": 0.048903508111834526, "clip_ratio/low_min": 0.048903508111834526, "clip_ratio/high_mean": 0.14840957894921303, "clip_ratio/high_max": 0.14840957894921303, "clip_ratio/region_mean": 0.19731308706104755, "reward_total_mean": 0.9933522939682007, "reward_meter_mean": 0.9933522939682007, "reward_meter_std": 0.004396223928779364, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9933522939682007, "reward_total_composite_std": 0.004396223928779364, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 52.0} {"timestamp_utc": "2026-04-11T19:29:01Z", "mode": "train", "global_step": 53, "epoch": 0.0020466481309854806, "loss": 0.0962, "grad_norm": 19.141048431396484, "learning_rate": 9.842424242424243e-06, "num_tokens": 105125.0, "completions/mean_length": 37.125, "completions/min_length": 29.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.125, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.5984563231468201, "rewards/meter/std": 0.42879587411880493, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5976195335388184, "rewards/total_composite/std": 0.43012017011642456, "reward": 0.5976195335388184, "reward_std": 0.43012017011642456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2187996506690979, "sampling/sampling_logp_difference/max": 1.363368034362793, "sampling/importance_sampling_ratio/min": 0.255797803401947, "sampling/importance_sampling_ratio/mean": 1.0220152139663696, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6774490028619766, "clip_ratio/low_mean": 0.06783267110586166, "clip_ratio/low_min": 0.06783267110586166, "clip_ratio/high_mean": 0.1322573497891426, "clip_ratio/high_max": 0.1322573497891426, "clip_ratio/region_mean": 0.20009002089500427, "reward_total_mean": 0.5976195335388184, "reward_meter_mean": 0.5984563231468201, "reward_meter_std": 0.42879587411880493, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5976195335388184, "reward_total_composite_std": 0.43012017011642456, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 53.0} {"timestamp_utc": "2026-04-11T19:29:06Z", "mode": "train", "global_step": 54, "epoch": 0.0020852641334569047, "loss": -0.066, "grad_norm": 15.735599517822266, "learning_rate": 9.83939393939394e-06, "num_tokens": 106862.0, "completions/mean_length": 52.125, "completions/min_length": 36.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.6028301119804382, "rewards/meter/std": 0.4117062985897064, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6028301119804382, "rewards/total_composite/std": 0.4117062985897064, "reward": 0.6028301119804382, "reward_std": 0.41170626878738403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21817506849765778, "sampling/sampling_logp_difference/max": 1.640699863433838, "sampling/importance_sampling_ratio/min": 0.19384431838989258, "sampling/importance_sampling_ratio/mean": 1.0488487482070923, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4166207313537598, "clip_ratio/low_mean": 0.08917682990431786, "clip_ratio/low_min": 0.08917682990431786, "clip_ratio/high_mean": 0.12507456727325916, "clip_ratio/high_max": 0.12507456727325916, "clip_ratio/region_mean": 0.21425139717757702, "reward_total_mean": 0.6028301119804382, "reward_meter_mean": 0.6028301119804382, "reward_meter_std": 0.4117062985897064, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6028301119804382, "reward_total_composite_std": 0.4117062985897064, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 54.0} {"timestamp_utc": "2026-04-11T19:29:11Z", "mode": "train", "global_step": 55, "epoch": 0.002123880135928329, "loss": 0.0736, "grad_norm": 16.236080169677734, "learning_rate": 9.836363636363637e-06, "num_tokens": 108605.0, "completions/mean_length": 57.875, "completions/min_length": 41.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.2679245173931122, "rewards/meter/std": 0.29882389307022095, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.2679245173931122, "rewards/total_composite/std": 0.29882389307022095, "reward": 0.2679245173931122, "reward_std": 0.29882386326789856, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2685631215572357, "sampling/sampling_logp_difference/max": 1.9843902587890625, "sampling/importance_sampling_ratio/min": 0.13746440410614014, "sampling/importance_sampling_ratio/mean": 1.0213844776153564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.436240643262863, "clip_ratio/low_mean": 0.1296012494713068, "clip_ratio/low_min": 0.1296012494713068, "clip_ratio/high_mean": 0.07919401116669178, "clip_ratio/high_max": 0.07919401116669178, "clip_ratio/region_mean": 0.20879526063799858, "reward_total_mean": 0.2679245173931122, "reward_meter_mean": 0.2679245173931122, "reward_meter_std": 0.29882389307022095, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.2679245173931122, "reward_total_composite_std": 0.29882389307022095, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 55.0} {"timestamp_utc": "2026-04-11T19:29:16Z", "mode": "train", "global_step": 56, "epoch": 0.002162496138399753, "loss": -0.1085, "grad_norm": 17.529335021972656, "learning_rate": 9.833333333333333e-06, "num_tokens": 110163.0, "completions/mean_length": 42.75, "completions/min_length": 24.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.7218412160873413, "rewards/meter/std": 0.4222549796104431, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6618984937667847, "rewards/total_composite/std": 0.41777893900871277, "reward": 0.6618984937667847, "reward_std": 0.4177789092063904, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20614919066429138, "sampling/sampling_logp_difference/max": 1.8073515892028809, "sampling/importance_sampling_ratio/min": 0.1972530037164688, "sampling/importance_sampling_ratio/mean": 1.0329205989837646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5369336754083633, "clip_ratio/low_mean": 0.054613095708191395, "clip_ratio/low_min": 0.054613095708191395, "clip_ratio/high_mean": 0.09804818406701088, "clip_ratio/high_max": 0.09804818406701088, "clip_ratio/region_mean": 0.15266127977520227, "reward_total_mean": 0.6618984937667847, "reward_meter_mean": 0.7218412160873413, "reward_meter_std": 0.4222549796104431, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6618984937667847, "reward_total_composite_std": 0.41777893900871277, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 56.0} {"timestamp_utc": "2026-04-11T19:29:23Z", "mode": "train", "global_step": 57, "epoch": 0.002201112140871177, "loss": 0.0427, "grad_norm": 7.2061991691589355, "learning_rate": 9.830303030303032e-06, "num_tokens": 113546.0, "completions/mean_length": 211.875, "completions/min_length": 156.0, "completions/max_length": 266.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 211.875, "completions/min_terminated_length": 156.0, "completions/max_terminated_length": 266.0, "rewards/meter/mean": 0.8386244773864746, "rewards/meter/std": 0.1593872308731079, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0578637570142746, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7174510955810547, "rewards/total_composite/std": 0.33248570561408997, "reward": 0.7174510955810547, "reward_std": 0.33248570561408997, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2227792590856552, "sampling/sampling_logp_difference/max": 1.7602176666259766, "sampling/importance_sampling_ratio/min": 0.17200742661952972, "sampling/importance_sampling_ratio/mean": 1.0385682582855225, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.097422182559967, "clip_ratio/low_mean": 0.07145614549517632, "clip_ratio/low_min": 0.07145614549517632, "clip_ratio/high_mean": 0.1381522510200739, "clip_ratio/high_max": 0.1381522510200739, "clip_ratio/region_mean": 0.2096083965152502, "reward_total_mean": 0.7174510955810547, "reward_meter_mean": 0.8386244773864746, "reward_meter_std": 0.1593872308731079, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0578637570142746, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7174510955810547, "reward_total_composite_std": 0.33248570561408997, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 57.0} {"timestamp_utc": "2026-04-11T19:29:28Z", "mode": "train", "global_step": 58, "epoch": 0.002239728143342601, "loss": 0.0085, "grad_norm": 10.99088191986084, "learning_rate": 9.827272727272729e-06, "num_tokens": 115430.0, "completions/mean_length": 66.5, "completions/min_length": 54.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8916192054748535, "rewards/meter/std": 0.292441725730896, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8916192054748535, "rewards/total_composite/std": 0.292441725730896, "reward": 0.8916192054748535, "reward_std": 0.292441725730896, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2055407166481018, "sampling/sampling_logp_difference/max": 2.950984001159668, "sampling/importance_sampling_ratio/min": 0.0522882305085659, "sampling/importance_sampling_ratio/mean": 1.0151267051696777, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.203368306159973, "clip_ratio/low_mean": 0.01875000074505806, "clip_ratio/low_min": 0.01875000074505806, "clip_ratio/high_mean": 0.20505854487419128, "clip_ratio/high_max": 0.20505854487419128, "clip_ratio/region_mean": 0.22380854561924934, "reward_total_mean": 0.8916192054748535, "reward_meter_mean": 0.8916192054748535, "reward_meter_std": 0.292441725730896, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8916192054748535, "reward_total_composite_std": 0.292441725730896, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 58.0} {"timestamp_utc": "2026-04-11T19:29:33Z", "mode": "train", "global_step": 59, "epoch": 0.002278344145814025, "loss": 0.1166, "grad_norm": 13.52698040008545, "learning_rate": 9.824242424242425e-06, "num_tokens": 117834.0, "completions/mean_length": 100.5, "completions/min_length": 88.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.5, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.30988091230392456, "rewards/meter/std": 0.22690658271312714, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3059839606285095, "rewards/total_composite/std": 0.22956949472427368, "reward": 0.3059839606285095, "reward_std": 0.2295694798231125, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2522209882736206, "sampling/sampling_logp_difference/max": 2.458967685699463, "sampling/importance_sampling_ratio/min": 0.08552319556474686, "sampling/importance_sampling_ratio/mean": 0.9983330368995667, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.076779752969742, "clip_ratio/low_mean": 0.13746320828795433, "clip_ratio/low_min": 0.13746320828795433, "clip_ratio/high_mean": 0.07533212564885616, "clip_ratio/high_max": 0.07533212564885616, "clip_ratio/region_mean": 0.2127953339368105, "reward_total_mean": 0.3059839606285095, "reward_meter_mean": 0.30988091230392456, "reward_meter_std": 0.22690658271312714, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3059839606285095, "reward_total_composite_std": 0.22956949472427368, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 59.0} {"timestamp_utc": "2026-04-11T19:29:38Z", "mode": "train", "global_step": 60, "epoch": 0.0023169601482854493, "loss": 0.0991, "grad_norm": 13.342586517333984, "learning_rate": 9.821212121212122e-06, "num_tokens": 119499.0, "completions/mean_length": 53.125, "completions/min_length": 34.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.3410363793373108, "rewards/meter/std": 0.28698158264160156, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3410363793373108, "rewards/total_composite/std": 0.28698158264160156, "reward": 0.3410363793373108, "reward_std": 0.28698158264160156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21170011162757874, "sampling/sampling_logp_difference/max": 1.3603332042694092, "sampling/importance_sampling_ratio/min": 0.2565752863883972, "sampling/importance_sampling_ratio/mean": 1.0315275192260742, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3633119463920593, "clip_ratio/low_mean": 0.12309817224740982, "clip_ratio/low_min": 0.12309817224740982, "clip_ratio/high_mean": 0.11208062618970871, "clip_ratio/high_max": 0.11208062618970871, "clip_ratio/region_mean": 0.23517879843711853, "reward_total_mean": 0.3410363793373108, "reward_meter_mean": 0.3410363793373108, "reward_meter_std": 0.28698158264160156, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3410363793373108, "reward_total_composite_std": 0.28698158264160156, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 60.0} {"timestamp_utc": "2026-04-11T19:29:42Z", "mode": "train", "global_step": 61, "epoch": 0.002355576150756874, "loss": 0.062, "grad_norm": 25.383638381958008, "learning_rate": 9.81818181818182e-06, "num_tokens": 121057.0, "completions/mean_length": 29.75, "completions/min_length": 20.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.75, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9133638143539429, "rewards/meter/std": 0.2053176313638687, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9133638143539429, "rewards/total_composite/std": 0.2053176313638687, "reward": 0.9133638143539429, "reward_std": 0.2053176462650299, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19405041635036469, "sampling/sampling_logp_difference/max": 1.3480195999145508, "sampling/importance_sampling_ratio/min": 0.25975415110588074, "sampling/importance_sampling_ratio/mean": 1.0374253988265991, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.489090621471405, "clip_ratio/low_mean": 0.036764707416296005, "clip_ratio/low_min": 0.036764707416296005, "clip_ratio/high_mean": 0.17636545840650797, "clip_ratio/high_max": 0.17636545840650797, "clip_ratio/region_mean": 0.21313016582280397, "reward_total_mean": 0.9133638143539429, "reward_meter_mean": 0.9133638143539429, "reward_meter_std": 0.2053176313638687, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9133638143539429, "reward_total_composite_std": 0.2053176313638687, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 61.0} {"timestamp_utc": "2026-04-11T19:29:47Z", "mode": "train", "global_step": 62, "epoch": 0.002394192153228298, "loss": 0.1325, "grad_norm": 12.204435348510742, "learning_rate": 9.815151515151516e-06, "num_tokens": 123204.0, "completions/mean_length": 99.375, "completions/min_length": 62.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.375, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.4881056547164917, "rewards/meter/std": 0.2547048330307007, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4881056547164917, "rewards/total_composite/std": 0.2547048330307007, "reward": 0.4881056547164917, "reward_std": 0.2547048032283783, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21773025393486023, "sampling/sampling_logp_difference/max": 1.875711441040039, "sampling/importance_sampling_ratio/min": 0.15324591100215912, "sampling/importance_sampling_ratio/mean": 1.041177749633789, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4795138239860535, "clip_ratio/low_mean": 0.100760068744421, "clip_ratio/low_min": 0.100760068744421, "clip_ratio/high_mean": 0.09878450445830822, "clip_ratio/high_max": 0.09878450445830822, "clip_ratio/region_mean": 0.19954457320272923, "reward_total_mean": 0.4881056547164917, "reward_meter_mean": 0.4881056547164917, "reward_meter_std": 0.2547048330307007, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4881056547164917, "reward_total_composite_std": 0.2547048330307007, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 62.0} {"timestamp_utc": "2026-04-11T19:29:52Z", "mode": "train", "global_step": 63, "epoch": 0.002432808155699722, "loss": 0.0373, "grad_norm": 18.67860221862793, "learning_rate": 9.812121212121212e-06, "num_tokens": 124939.0, "completions/mean_length": 60.875, "completions/min_length": 44.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9198144674301147, "rewards/meter/std": 0.13751091063022614, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9198144674301147, "rewards/total_composite/std": 0.13751091063022614, "reward": 0.9198144674301147, "reward_std": 0.13751091063022614, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19855335354804993, "sampling/sampling_logp_difference/max": 1.4531702995300293, "sampling/importance_sampling_ratio/min": 0.23382781445980072, "sampling/importance_sampling_ratio/mean": 1.0526179075241089, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3204995840787888, "clip_ratio/low_mean": 0.02020994247868657, "clip_ratio/low_min": 0.02020994247868657, "clip_ratio/high_mean": 0.1503895577043295, "clip_ratio/high_max": 0.1503895577043295, "clip_ratio/region_mean": 0.17059950018301606, "reward_total_mean": 0.9198144674301147, "reward_meter_mean": 0.9198144674301147, "reward_meter_std": 0.13751091063022614, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9198144674301147, "reward_total_composite_std": 0.13751091063022614, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 63.0} {"timestamp_utc": "2026-04-11T19:30:00Z", "mode": "train", "global_step": 64, "epoch": 0.002471424158171146, "loss": -0.052, "grad_norm": 5.2523932456970215, "learning_rate": 9.809090909090911e-06, "num_tokens": 129130.0, "completions/mean_length": 308.875, "completions/min_length": 248.0, "completions/max_length": 410.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 308.875, "completions/min_terminated_length": 248.0, "completions/max_terminated_length": 410.0, "rewards/meter/mean": 0.9247400164604187, "rewards/meter/std": 0.13669492304325104, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.05175492912530899, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.5670943260192871, "rewards/total_composite/std": 0.47056710720062256, "reward": 0.5670943260192871, "reward_std": 0.47056710720062256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21920810639858246, "sampling/sampling_logp_difference/max": 1.8417625427246094, "sampling/importance_sampling_ratio/min": 0.15853774547576904, "sampling/importance_sampling_ratio/mean": 1.0358939170837402, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1392965018749237, "clip_ratio/low_mean": 0.06851893290877342, "clip_ratio/low_min": 0.06851893290877342, "clip_ratio/high_mean": 0.1345098316669464, "clip_ratio/high_max": 0.1345098316669464, "clip_ratio/region_mean": 0.20302876457571983, "reward_total_mean": 0.5670943260192871, "reward_meter_mean": 0.9247400164604187, "reward_meter_std": 0.13669492304325104, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.05175492912530899, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.5670943260192871, "reward_total_composite_std": 0.47056710720062256, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 64.0} {"timestamp_utc": "2026-04-11T19:30:05Z", "mode": "train", "global_step": 65, "epoch": 0.0025100401606425703, "loss": 0.2664, "grad_norm": 26.357711791992188, "learning_rate": 9.806060606060607e-06, "num_tokens": 130721.0, "completions/mean_length": 35.875, "completions/min_length": 22.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.875, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.5192750096321106, "rewards/meter/std": 0.47312480211257935, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5192750096321106, "rewards/total_composite/std": 0.47312480211257935, "reward": 0.5192750096321106, "reward_std": 0.47312480211257935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2368556261062622, "sampling/sampling_logp_difference/max": 2.032594680786133, "sampling/importance_sampling_ratio/min": 0.13099518418312073, "sampling/importance_sampling_ratio/mean": 1.0055742263793945, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4370453506708145, "clip_ratio/low_mean": 0.11948978900909424, "clip_ratio/low_min": 0.11948978900909424, "clip_ratio/high_mean": 0.08198772463947535, "clip_ratio/high_max": 0.08198772463947535, "clip_ratio/region_mean": 0.20147751364856958, "reward_total_mean": 0.5192750096321106, "reward_meter_mean": 0.5192750096321106, "reward_meter_std": 0.47312480211257935, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5192750096321106, "reward_total_composite_std": 0.47312480211257935, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 65.0} {"timestamp_utc": "2026-04-11T19:30:09Z", "mode": "train", "global_step": 66, "epoch": 0.0025486561631139944, "loss": 0.137, "grad_norm": 27.630306243896484, "learning_rate": 9.803030303030304e-06, "num_tokens": 132488.0, "completions/mean_length": 66.875, "completions/min_length": 44.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.7682692408561707, "rewards/meter/std": 0.3781956434249878, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7682692408561707, "rewards/total_composite/std": 0.3781956434249878, "reward": 0.7682692408561707, "reward_std": 0.3781956434249878, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21417810022830963, "sampling/sampling_logp_difference/max": 2.0331544876098633, "sampling/importance_sampling_ratio/min": 0.1309218853712082, "sampling/importance_sampling_ratio/mean": 1.000638723373413, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.377866268157959, "clip_ratio/low_mean": 0.046164773404598236, "clip_ratio/low_min": 0.046164773404598236, "clip_ratio/high_mean": 0.13170645385980606, "clip_ratio/high_max": 0.13170645385980606, "clip_ratio/region_mean": 0.1778712272644043, "reward_total_mean": 0.7682692408561707, "reward_meter_mean": 0.7682692408561707, "reward_meter_std": 0.3781956434249878, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7682692408561707, "reward_total_composite_std": 0.3781956434249878, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 66.0} {"timestamp_utc": "2026-04-11T19:30:14Z", "mode": "train", "global_step": 67, "epoch": 0.0025872721655854185, "loss": 0.0209, "grad_norm": 14.999178886413574, "learning_rate": 9.800000000000001e-06, "num_tokens": 134295.0, "completions/mean_length": 48.875, "completions/min_length": 41.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8183262348175049, "rewards/meter/std": 0.32211264967918396, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6665328741073608, "rewards/total_composite/std": 0.43555328249931335, "reward": 0.6665328741073608, "reward_std": 0.43555328249931335, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2460046112537384, "sampling/sampling_logp_difference/max": 1.4428062438964844, "sampling/importance_sampling_ratio/min": 0.23626382648944855, "sampling/importance_sampling_ratio/mean": 1.0471768379211426, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.8289676904678345, "clip_ratio/low_mean": 0.08497239649295807, "clip_ratio/low_min": 0.08497239649295807, "clip_ratio/high_mean": 0.13258693367242813, "clip_ratio/high_max": 0.13258693367242813, "clip_ratio/region_mean": 0.2175593301653862, "reward_total_mean": 0.6665328741073608, "reward_meter_mean": 0.8183262348175049, "reward_meter_std": 0.32211264967918396, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6665328741073608, "reward_total_composite_std": 0.43555328249931335, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 67.0} {"timestamp_utc": "2026-04-11T19:30:18Z", "mode": "train", "global_step": 68, "epoch": 0.0026258881680568426, "loss": 0.1484, "grad_norm": 15.564022064208984, "learning_rate": 9.796969696969698e-06, "num_tokens": 135997.0, "completions/mean_length": 52.75, "completions/min_length": 43.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.4771120548248291, "rewards/meter/std": 0.3203728497028351, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.35390323400497437, "rewards/total_composite/std": 0.28436195850372314, "reward": 0.35390323400497437, "reward_std": 0.28436198830604553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22163131833076477, "sampling/sampling_logp_difference/max": 1.4564504623413086, "sampling/importance_sampling_ratio/min": 0.23306208848953247, "sampling/importance_sampling_ratio/mean": 1.0302598476409912, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6037258356809616, "clip_ratio/low_mean": 0.08730671741068363, "clip_ratio/low_min": 0.08730671741068363, "clip_ratio/high_mean": 0.10875347442924976, "clip_ratio/high_max": 0.10875347442924976, "clip_ratio/region_mean": 0.1960601918399334, "reward_total_mean": 0.35390323400497437, "reward_meter_mean": 0.4771120548248291, "reward_meter_std": 0.3203728497028351, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.35390323400497437, "reward_total_composite_std": 0.28436195850372314, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 68.0} {"timestamp_utc": "2026-04-11T19:30:24Z", "mode": "train", "global_step": 69, "epoch": 0.002664504170528267, "loss": 0.1465, "grad_norm": 8.40172290802002, "learning_rate": 9.793939393939394e-06, "num_tokens": 138233.0, "completions/mean_length": 112.5, "completions/min_length": 77.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.5, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.6559433937072754, "rewards/meter/std": 0.3531401455402374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5694236159324646, "rewards/total_composite/std": 0.4212262034416199, "reward": 0.5694236159324646, "reward_std": 0.4212261736392975, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21621394157409668, "sampling/sampling_logp_difference/max": 1.6637945175170898, "sampling/importance_sampling_ratio/min": 0.18941885232925415, "sampling/importance_sampling_ratio/mean": 1.0383771657943726, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.092645615339279, "clip_ratio/low_mean": 0.06984523870050907, "clip_ratio/low_min": 0.06984523870050907, "clip_ratio/high_mean": 0.15749920904636383, "clip_ratio/high_max": 0.15749920904636383, "clip_ratio/region_mean": 0.2273444477468729, "reward_total_mean": 0.5694236159324646, "reward_meter_mean": 0.6559433937072754, "reward_meter_std": 0.3531401455402374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5694236159324646, "reward_total_composite_std": 0.4212262034416199, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 69.0} {"timestamp_utc": "2026-04-11T19:30:29Z", "mode": "train", "global_step": 70, "epoch": 0.0027031201729996912, "loss": 0.0207, "grad_norm": 12.81651782989502, "learning_rate": 9.790909090909093e-06, "num_tokens": 140310.0, "completions/mean_length": 78.625, "completions/min_length": 64.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.625, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.43477553129196167, "rewards/meter/std": 0.332572340965271, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4240095019340515, "rewards/total_composite/std": 0.337272584438324, "reward": 0.4240095019340515, "reward_std": 0.337272584438324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2150382101535797, "sampling/sampling_logp_difference/max": 1.2891788482666016, "sampling/importance_sampling_ratio/min": 0.2754969000816345, "sampling/importance_sampling_ratio/mean": 1.031362533569336, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6834762394428253, "clip_ratio/low_mean": 0.10936851240694523, "clip_ratio/low_min": 0.10936851240694523, "clip_ratio/high_mean": 0.07498700357973576, "clip_ratio/high_max": 0.07498700357973576, "clip_ratio/region_mean": 0.18435551598668098, "reward_total_mean": 0.4240095019340515, "reward_meter_mean": 0.43477553129196167, "reward_meter_std": 0.332572340965271, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4240095019340515, "reward_total_composite_std": 0.337272584438324, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 70.0} {"timestamp_utc": "2026-04-11T19:30:34Z", "mode": "train", "global_step": 71, "epoch": 0.0027417361754711153, "loss": 0.0109, "grad_norm": 13.432793617248535, "learning_rate": 9.787878787878788e-06, "num_tokens": 142298.0, "completions/mean_length": 77.5, "completions/min_length": 47.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.7772670984268188, "rewards/meter/std": 0.3389206826686859, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7772670984268188, "rewards/total_composite/std": 0.3389206826686859, "reward": 0.7772670984268188, "reward_std": 0.3389207124710083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1868724375963211, "sampling/sampling_logp_difference/max": 1.5989351272583008, "sampling/importance_sampling_ratio/min": 0.2021116316318512, "sampling/importance_sampling_ratio/mean": 1.0325590372085571, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1531532555818558, "clip_ratio/low_mean": 0.07749529182910919, "clip_ratio/low_min": 0.07749529182910919, "clip_ratio/high_mean": 0.12004932574927807, "clip_ratio/high_max": 0.12004932574927807, "clip_ratio/region_mean": 0.19754461757838726, "reward_total_mean": 0.7772670984268188, "reward_meter_mean": 0.7772670984268188, "reward_meter_std": 0.3389206826686859, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7772670984268188, "reward_total_composite_std": 0.3389206826686859, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 71.0} {"timestamp_utc": "2026-04-11T19:30:41Z", "mode": "train", "global_step": 72, "epoch": 0.0027803521779425394, "loss": 0.1169, "grad_norm": 9.057010650634766, "learning_rate": 9.784848484848486e-06, "num_tokens": 144949.0, "completions/mean_length": 153.375, "completions/min_length": 138.0, "completions/max_length": 175.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 153.375, "completions/min_terminated_length": 138.0, "completions/max_terminated_length": 175.0, "rewards/meter/mean": 0.5224494934082031, "rewards/meter/std": 0.4397116005420685, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.1414213478565216, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5220509767532349, "rewards/total_composite/std": 0.4402455985546112, "reward": 0.5220509767532349, "reward_std": 0.4402455985546112, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2253074198961258, "sampling/sampling_logp_difference/max": 1.7899761199951172, "sampling/importance_sampling_ratio/min": 0.16696415841579437, "sampling/importance_sampling_ratio/mean": 1.0310229063034058, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.355428859591484, "clip_ratio/low_mean": 0.0991472564637661, "clip_ratio/low_min": 0.0991472564637661, "clip_ratio/high_mean": 0.11257654242217541, "clip_ratio/high_max": 0.11257654242217541, "clip_ratio/region_mean": 0.2117237988859415, "reward_total_mean": 0.5220509767532349, "reward_meter_mean": 0.5224494934082031, "reward_meter_std": 0.4397116005420685, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.1414213478565216, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5220509767532349, "reward_total_composite_std": 0.4402455985546112, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 72.0} {"timestamp_utc": "2026-04-11T19:30:46Z", "mode": "train", "global_step": 73, "epoch": 0.0028189681804139635, "loss": 0.025, "grad_norm": 14.417825698852539, "learning_rate": 9.781818181818183e-06, "num_tokens": 146868.0, "completions/mean_length": 56.875, "completions/min_length": 47.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.44528478384017944, "rewards/meter/std": 0.402529776096344, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.40673646330833435, "rewards/total_composite/std": 0.43125420808792114, "reward": 0.40673646330833435, "reward_std": 0.43125417828559875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2235013246536255, "sampling/sampling_logp_difference/max": 1.7862257957458496, "sampling/importance_sampling_ratio/min": 0.16759149730205536, "sampling/importance_sampling_ratio/mean": 1.008949637413025, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.409619852900505, "clip_ratio/low_mean": 0.10295584239065647, "clip_ratio/low_min": 0.10295584239065647, "clip_ratio/high_mean": 0.11561089940369129, "clip_ratio/high_max": 0.11561089940369129, "clip_ratio/region_mean": 0.21856674179434776, "reward_total_mean": 0.40673646330833435, "reward_meter_mean": 0.44528478384017944, "reward_meter_std": 0.402529776096344, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.40673646330833435, "reward_total_composite_std": 0.43125420808792114, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 73.0} {"timestamp_utc": "2026-04-11T19:30:56Z", "mode": "train", "global_step": 74, "epoch": 0.0028575841828853876, "loss": -0.011, "grad_norm": 1.7615196704864502, "learning_rate": 9.77878787878788e-06, "num_tokens": 150527.0, "completions/mean_length": 479.375, "completions/min_length": 426.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 446.75, "completions/min_terminated_length": 426.0, "completions/max_terminated_length": 465.0, "rewards/meter/mean": 0.6778547763824463, "rewards/meter/std": 0.2513851225376129, "rewards/count_adherence/mean": 0.8833333253860474, "rewards/count_adherence/std": 0.077664315700531, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5821076035499573, "rewards/total_composite/std": 0.29507479071617126, "reward": 0.5821076035499573, "reward_std": 0.29507479071617126, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22136430442333221, "sampling/sampling_logp_difference/max": 1.71954345703125, "sampling/importance_sampling_ratio/min": 0.17914791405200958, "sampling/importance_sampling_ratio/mean": 1.039528250694275, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.72447469830513, "clip_ratio/low_mean": 0.05759216099977493, "clip_ratio/low_min": 0.05759216099977493, "clip_ratio/high_mean": 0.02222863771021366, "clip_ratio/high_max": 0.02222863771021366, "clip_ratio/region_mean": 0.0798207987099886, "reward_total_mean": 0.5821076035499573, "reward_meter_mean": 0.6778547763824463, "reward_meter_std": 0.2513851225376129, "reward_count_adherence_mean": 0.8833333253860474, "reward_count_adherence_std": 0.077664315700531, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5821076035499573, "reward_total_composite_std": 0.29507479071617126, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 74.0} {"timestamp_utc": "2026-04-11T19:31:02Z", "mode": "train", "global_step": 75, "epoch": 0.0028962001853568117, "loss": 0.0601, "grad_norm": 10.22510051727295, "learning_rate": 9.775757575757576e-06, "num_tokens": 152828.0, "completions/mean_length": 116.625, "completions/min_length": 87.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.625, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.449047327041626, "rewards/meter/std": 0.3430718183517456, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.43358057737350464, "rewards/total_composite/std": 0.3434963822364807, "reward": 0.43358057737350464, "reward_std": 0.3434963822364807, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22043807804584503, "sampling/sampling_logp_difference/max": 1.9572572708129883, "sampling/importance_sampling_ratio/min": 0.1412452757358551, "sampling/importance_sampling_ratio/mean": 1.027030110359192, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2147320955991745, "clip_ratio/low_mean": 0.11534719914197922, "clip_ratio/low_min": 0.11534719914197922, "clip_ratio/high_mean": 0.08039004355669022, "clip_ratio/high_max": 0.08039004355669022, "clip_ratio/region_mean": 0.19573724269866943, "reward_total_mean": 0.43358057737350464, "reward_meter_mean": 0.449047327041626, "reward_meter_std": 0.3430718183517456, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.43358057737350464, "reward_total_composite_std": 0.3434963822364807, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 75.0} {"timestamp_utc": "2026-04-11T19:31:07Z", "mode": "train", "global_step": 76, "epoch": 0.002934816187828236, "loss": 0.126, "grad_norm": 14.32852840423584, "learning_rate": 9.772727272727273e-06, "num_tokens": 154686.0, "completions/mean_length": 55.25, "completions/min_length": 38.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.7078003883361816, "rewards/meter/std": 0.4165555238723755, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7078003883361816, "rewards/total_composite/std": 0.4165555238723755, "reward": 0.7078003883361816, "reward_std": 0.4165555238723755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22555997967720032, "sampling/sampling_logp_difference/max": 1.839177131652832, "sampling/importance_sampling_ratio/min": 0.15894815325737, "sampling/importance_sampling_ratio/mean": 1.0568245649337769, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0069815814495087, "clip_ratio/low_mean": 0.09451080299913883, "clip_ratio/low_min": 0.09451080299913883, "clip_ratio/high_mean": 0.12531227804720402, "clip_ratio/high_max": 0.12531227804720402, "clip_ratio/region_mean": 0.21982308104634285, "reward_total_mean": 0.7078003883361816, "reward_meter_mean": 0.7078003883361816, "reward_meter_std": 0.4165555238723755, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7078003883361816, "reward_total_composite_std": 0.4165555238723755, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 76.0} {"timestamp_utc": "2026-04-11T19:31:12Z", "mode": "train", "global_step": 77, "epoch": 0.0029734321902996604, "loss": 0.0662, "grad_norm": 16.326507568359375, "learning_rate": 9.76969696969697e-06, "num_tokens": 156293.0, "completions/mean_length": 48.875, "completions/min_length": 34.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.6651355624198914, "rewards/meter/std": 0.39970725774765015, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6651355624198914, "rewards/total_composite/std": 0.39970725774765015, "reward": 0.6651355624198914, "reward_std": 0.39970725774765015, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22144979238510132, "sampling/sampling_logp_difference/max": 1.6073646545410156, "sampling/importance_sampling_ratio/min": 0.20041508972644806, "sampling/importance_sampling_ratio/mean": 1.023947834968567, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4677112847566605, "clip_ratio/low_mean": 0.08010341972112656, "clip_ratio/low_min": 0.08010341972112656, "clip_ratio/high_mean": 0.15934639982879162, "clip_ratio/high_max": 0.15934639982879162, "clip_ratio/region_mean": 0.23944981954991817, "reward_total_mean": 0.6651355624198914, "reward_meter_mean": 0.6651355624198914, "reward_meter_std": 0.39970725774765015, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6651355624198914, "reward_total_composite_std": 0.39970725774765015, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 77.0} {"timestamp_utc": "2026-04-11T19:31:19Z", "mode": "train", "global_step": 78, "epoch": 0.0030120481927710845, "loss": 0.073, "grad_norm": 8.249039649963379, "learning_rate": 9.766666666666667e-06, "num_tokens": 159067.0, "completions/mean_length": 148.75, "completions/min_length": 127.0, "completions/max_length": 197.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 148.75, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 197.0, "rewards/meter/mean": 0.3600383400917053, "rewards/meter/std": 0.2478063702583313, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.35600364208221436, "rewards/total_composite/std": 0.2517344057559967, "reward": 0.35600364208221436, "reward_std": 0.2517344057559967, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20435652136802673, "sampling/sampling_logp_difference/max": 1.835897445678711, "sampling/importance_sampling_ratio/min": 0.1594703197479248, "sampling/importance_sampling_ratio/mean": 1.0375880002975464, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.010555699467659, "clip_ratio/low_mean": 0.1108522079885006, "clip_ratio/low_min": 0.1108522079885006, "clip_ratio/high_mean": 0.06962481327354908, "clip_ratio/high_max": 0.06962481327354908, "clip_ratio/region_mean": 0.18047702126204967, "reward_total_mean": 0.35600364208221436, "reward_meter_mean": 0.3600383400917053, "reward_meter_std": 0.2478063702583313, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.35600364208221436, "reward_total_composite_std": 0.2517344057559967, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 78.0} {"timestamp_utc": "2026-04-11T19:31:27Z", "mode": "train", "global_step": 79, "epoch": 0.0030506641952425086, "loss": -0.0364, "grad_norm": 5.221505165100098, "learning_rate": 9.763636363636365e-06, "num_tokens": 163339.0, "completions/mean_length": 315.0, "completions/min_length": 211.0, "completions/max_length": 384.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 315.0, "completions/min_terminated_length": 211.0, "completions/max_terminated_length": 384.0, "rewards/meter/mean": 0.4919300973415375, "rewards/meter/std": 0.2810616195201874, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.10690447688102722, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4219393730163574, "rewards/total_composite/std": 0.2564575970172882, "reward": 0.4219393730163574, "reward_std": 0.2564575970172882, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20352312922477722, "sampling/sampling_logp_difference/max": 1.9669795036315918, "sampling/importance_sampling_ratio/min": 0.13987872004508972, "sampling/importance_sampling_ratio/mean": 1.0315507650375366, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.524102747440338, "clip_ratio/low_mean": 0.11126292496919632, "clip_ratio/low_min": 0.11126292496919632, "clip_ratio/high_mean": 0.06808187626302242, "clip_ratio/high_max": 0.06808187626302242, "clip_ratio/region_mean": 0.17934480123221874, "reward_total_mean": 0.4219393730163574, "reward_meter_mean": 0.4919300973415375, "reward_meter_std": 0.2810616195201874, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.10690447688102722, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4219393730163574, "reward_total_composite_std": 0.2564575970172882, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 79.0} {"timestamp_utc": "2026-04-11T19:31:32Z", "mode": "train", "global_step": 80, "epoch": 0.0030892801977139327, "loss": 0.0304, "grad_norm": 15.532655715942383, "learning_rate": 9.760606060606062e-06, "num_tokens": 165053.0, "completions/mean_length": 55.25, "completions/min_length": 39.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7070286273956299, "rewards/meter/std": 0.3917587697505951, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7070286273956299, "rewards/total_composite/std": 0.3917587697505951, "reward": 0.7070286273956299, "reward_std": 0.3917587697505951, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2240315079689026, "sampling/sampling_logp_difference/max": 1.0939302444458008, "sampling/importance_sampling_ratio/min": 0.3348976671695709, "sampling/importance_sampling_ratio/mean": 1.0330888032913208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6958227306604385, "clip_ratio/low_mean": 0.05844542942941189, "clip_ratio/low_min": 0.05844542942941189, "clip_ratio/high_mean": 0.1331654218956828, "clip_ratio/high_max": 0.1331654218956828, "clip_ratio/region_mean": 0.1916108513250947, "reward_total_mean": 0.7070286273956299, "reward_meter_mean": 0.7070286273956299, "reward_meter_std": 0.3917587697505951, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7070286273956299, "reward_total_composite_std": 0.3917587697505951, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 80.0} {"timestamp_utc": "2026-04-11T19:31:36Z", "mode": "train", "global_step": 81, "epoch": 0.003127896200185357, "loss": -0.0475, "grad_norm": 20.511795043945312, "learning_rate": 9.757575757575758e-06, "num_tokens": 166596.0, "completions/mean_length": 40.875, "completions/min_length": 28.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6753016710281372, "rewards/meter/std": 0.3897930681705475, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6753016710281372, "rewards/total_composite/std": 0.3897930681705475, "reward": 0.6753016710281372, "reward_std": 0.3897930681705475, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17763756215572357, "sampling/sampling_logp_difference/max": 2.181588649749756, "sampling/importance_sampling_ratio/min": 0.11286209523677826, "sampling/importance_sampling_ratio/mean": 0.9999099969863892, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8825555518269539, "clip_ratio/low_mean": 0.05491071380674839, "clip_ratio/low_min": 0.05491071380674839, "clip_ratio/high_mean": 0.14205198176205158, "clip_ratio/high_max": 0.14205198176205158, "clip_ratio/region_mean": 0.19696269556879997, "reward_total_mean": 0.6753016710281372, "reward_meter_mean": 0.6753016710281372, "reward_meter_std": 0.3897930681705475, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6753016710281372, "reward_total_composite_std": 0.3897930681705475, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 81.0} {"timestamp_utc": "2026-04-11T19:31:42Z", "mode": "train", "global_step": 82, "epoch": 0.003166512202656781, "loss": 0.1511, "grad_norm": 11.253849029541016, "learning_rate": 9.754545454545455e-06, "num_tokens": 168602.0, "completions/mean_length": 74.75, "completions/min_length": 57.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.7240955233573914, "rewards/meter/std": 0.410367876291275, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6087316870689392, "rewards/total_composite/std": 0.4716724455356598, "reward": 0.6087316870689392, "reward_std": 0.4716724455356598, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2147466540336609, "sampling/sampling_logp_difference/max": 1.7266242504119873, "sampling/importance_sampling_ratio/min": 0.17788389325141907, "sampling/importance_sampling_ratio/mean": 1.0235908031463623, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4969915598630905, "clip_ratio/low_mean": 0.07035921700298786, "clip_ratio/low_min": 0.07035921700298786, "clip_ratio/high_mean": 0.13331562653183937, "clip_ratio/high_max": 0.13331562653183937, "clip_ratio/region_mean": 0.20367484353482723, "reward_total_mean": 0.6087316870689392, "reward_meter_mean": 0.7240955233573914, "reward_meter_std": 0.410367876291275, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6087316870689392, "reward_total_composite_std": 0.4716724455356598, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 82.0} {"timestamp_utc": "2026-04-11T19:31:49Z", "mode": "train", "global_step": 83, "epoch": 0.003205128205128205, "loss": -0.0125, "grad_norm": 7.257972240447998, "learning_rate": 9.751515151515152e-06, "num_tokens": 171482.0, "completions/mean_length": 179.0, "completions/min_length": 100.0, "completions/max_length": 264.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 179.0, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 264.0, "rewards/meter/mean": 0.5834420323371887, "rewards/meter/std": 0.35952043533325195, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.4665340185165405, "rewards/total_composite/std": 0.38863879442214966, "reward": 0.4665340185165405, "reward_std": 0.38863876461982727, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20953215658664703, "sampling/sampling_logp_difference/max": 2.045928955078125, "sampling/importance_sampling_ratio/min": 0.12926006317138672, "sampling/importance_sampling_ratio/mean": 1.048352599143982, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.964230954647064, "clip_ratio/low_mean": 0.09328041970729828, "clip_ratio/low_min": 0.09328041970729828, "clip_ratio/high_mean": 0.10117011703550816, "clip_ratio/high_max": 0.10117011703550816, "clip_ratio/region_mean": 0.19445053674280643, "reward_total_mean": 0.4665340185165405, "reward_meter_mean": 0.5834420323371887, "reward_meter_std": 0.35952043533325195, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.4665340185165405, "reward_total_composite_std": 0.38863879442214966, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 83.0} {"timestamp_utc": "2026-04-11T19:31:54Z", "mode": "train", "global_step": 84, "epoch": 0.003243744207599629, "loss": 0.1851, "grad_norm": 20.633411407470703, "learning_rate": 9.74848484848485e-06, "num_tokens": 173072.0, "completions/mean_length": 51.75, "completions/min_length": 33.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.36928117275238037, "rewards/meter/std": 0.413564532995224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.36928117275238037, "rewards/total_composite/std": 0.413564532995224, "reward": 0.36928117275238037, "reward_std": 0.413564532995224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20136907696723938, "sampling/sampling_logp_difference/max": 1.7371244430541992, "sampling/importance_sampling_ratio/min": 0.176025852560997, "sampling/importance_sampling_ratio/mean": 1.0323539972305298, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0957568138837814, "clip_ratio/low_mean": 0.10067356191575527, "clip_ratio/low_min": 0.10067356191575527, "clip_ratio/high_mean": 0.06479034759104252, "clip_ratio/high_max": 0.06479034759104252, "clip_ratio/region_mean": 0.1654639095067978, "reward_total_mean": 0.36928117275238037, "reward_meter_mean": 0.36928117275238037, "reward_meter_std": 0.413564532995224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.36928117275238037, "reward_total_composite_std": 0.413564532995224, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 84.0} {"timestamp_utc": "2026-04-11T19:32:00Z", "mode": "train", "global_step": 85, "epoch": 0.0032823602100710537, "loss": -0.0541, "grad_norm": 9.771239280700684, "learning_rate": 9.745454545454547e-06, "num_tokens": 175025.0, "completions/mean_length": 90.125, "completions/min_length": 56.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.125, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.6992344260215759, "rewards/meter/std": 0.35877659916877747, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6992344260215759, "rewards/total_composite/std": 0.35877659916877747, "reward": 0.6992344260215759, "reward_std": 0.3587765693664551, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22157631814479828, "sampling/sampling_logp_difference/max": 1.8856267929077148, "sampling/importance_sampling_ratio/min": 0.15173391997814178, "sampling/importance_sampling_ratio/mean": 1.0352109670639038, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0091913044452667, "clip_ratio/low_mean": 0.0852907095104456, "clip_ratio/low_min": 0.0852907095104456, "clip_ratio/high_mean": 0.10693838447332382, "clip_ratio/high_max": 0.10693838447332382, "clip_ratio/region_mean": 0.19222909398376942, "reward_total_mean": 0.6992344260215759, "reward_meter_mean": 0.6992344260215759, "reward_meter_std": 0.35877659916877747, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6992344260215759, "reward_total_composite_std": 0.35877659916877747, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 85.0} {"timestamp_utc": "2026-04-11T19:32:06Z", "mode": "train", "global_step": 86, "epoch": 0.0033209762125424778, "loss": 0.0244, "grad_norm": 6.172459125518799, "learning_rate": 9.742424242424244e-06, "num_tokens": 178525.0, "completions/mean_length": 230.5, "completions/min_length": 198.0, "completions/max_length": 254.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 230.5, "completions/min_terminated_length": 198.0, "completions/max_terminated_length": 254.0, "rewards/meter/mean": 0.2989215552806854, "rewards/meter/std": 0.2961769700050354, "rewards/count_adherence/mean": 0.9821428656578064, "rewards/count_adherence/std": 0.05050762742757797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.2985811233520508, "rewards/total_composite/std": 0.29654595255851746, "reward": 0.2985811233520508, "reward_std": 0.29654595255851746, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20698504149913788, "sampling/sampling_logp_difference/max": 2.3154473304748535, "sampling/importance_sampling_ratio/min": 0.09872201830148697, "sampling/importance_sampling_ratio/mean": 1.0293720960617065, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.359074369072914, "clip_ratio/low_mean": 0.13297666609287262, "clip_ratio/low_min": 0.13297666609287262, "clip_ratio/high_mean": 0.05019771121442318, "clip_ratio/high_max": 0.05019771121442318, "clip_ratio/region_mean": 0.1831743773072958, "reward_total_mean": 0.2985811233520508, "reward_meter_mean": 0.2989215552806854, "reward_meter_std": 0.2961769700050354, "reward_count_adherence_mean": 0.9821428656578064, "reward_count_adherence_std": 0.05050762742757797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.2985811233520508, "reward_total_composite_std": 0.29654595255851746, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 86.0} {"timestamp_utc": "2026-04-11T19:32:11Z", "mode": "train", "global_step": 87, "epoch": 0.003359592215013902, "loss": 0.049, "grad_norm": 16.136564254760742, "learning_rate": 9.739393939393941e-06, "num_tokens": 180121.0, "completions/mean_length": 45.5, "completions/min_length": 33.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.5413487553596497, "rewards/meter/std": 0.41876327991485596, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5413487553596497, "rewards/total_composite/std": 0.41876327991485596, "reward": 0.5413487553596497, "reward_std": 0.41876330971717834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20200753211975098, "sampling/sampling_logp_difference/max": 1.7855415344238281, "sampling/importance_sampling_ratio/min": 0.1677062213420868, "sampling/importance_sampling_ratio/mean": 1.0214216709136963, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9420476108789444, "clip_ratio/low_mean": 0.07651033625006676, "clip_ratio/low_min": 0.07651033625006676, "clip_ratio/high_mean": 0.09015538915991783, "clip_ratio/high_max": 0.09015538915991783, "clip_ratio/region_mean": 0.1666657254099846, "reward_total_mean": 0.5413487553596497, "reward_meter_mean": 0.5413487553596497, "reward_meter_std": 0.41876327991485596, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5413487553596497, "reward_total_composite_std": 0.41876327991485596, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 87.0} {"timestamp_utc": "2026-04-11T19:32:17Z", "mode": "train", "global_step": 88, "epoch": 0.003398208217485326, "loss": 0.0501, "grad_norm": 7.712515830993652, "learning_rate": 9.736363636363637e-06, "num_tokens": 183183.0, "completions/mean_length": 171.75, "completions/min_length": 156.0, "completions/max_length": 196.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 171.75, "completions/min_terminated_length": 156.0, "completions/max_terminated_length": 196.0, "rewards/meter/mean": 0.8209783434867859, "rewards/meter/std": 0.20396688580513, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.773547887802124, "rewards/total_composite/std": 0.18800683319568634, "reward": 0.773547887802124, "reward_std": 0.18800680339336395, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17306587100028992, "sampling/sampling_logp_difference/max": 1.458749771118164, "sampling/importance_sampling_ratio/min": 0.23252680897712708, "sampling/importance_sampling_ratio/mean": 1.0262449979782104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.721228078007698, "clip_ratio/low_mean": 0.0701658520847559, "clip_ratio/low_min": 0.0701658520847559, "clip_ratio/high_mean": 0.09816804714500904, "clip_ratio/high_max": 0.09816804714500904, "clip_ratio/region_mean": 0.16833389922976494, "reward_total_mean": 0.773547887802124, "reward_meter_mean": 0.8209783434867859, "reward_meter_std": 0.20396688580513, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.773547887802124, "reward_total_composite_std": 0.18800683319568634, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 88.0} {"timestamp_utc": "2026-04-11T19:32:22Z", "mode": "train", "global_step": 89, "epoch": 0.00343682421995675, "loss": 0.0546, "grad_norm": 11.22130298614502, "learning_rate": 9.733333333333334e-06, "num_tokens": 185260.0, "completions/mean_length": 88.625, "completions/min_length": 73.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.625, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.6381306648254395, "rewards/meter/std": 0.44584599137306213, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6381306648254395, "rewards/total_composite/std": 0.44584599137306213, "reward": 0.6381306648254395, "reward_std": 0.44584596157073975, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19463911652565002, "sampling/sampling_logp_difference/max": 2.2131471633911133, "sampling/importance_sampling_ratio/min": 0.10935594886541367, "sampling/importance_sampling_ratio/mean": 1.0421054363250732, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.467219904065132, "clip_ratio/low_mean": 0.07750886678695679, "clip_ratio/low_min": 0.07750886678695679, "clip_ratio/high_mean": 0.12146328948438168, "clip_ratio/high_max": 0.12146328948438168, "clip_ratio/region_mean": 0.19897215627133846, "reward_total_mean": 0.6381306648254395, "reward_meter_mean": 0.6381306648254395, "reward_meter_std": 0.44584599137306213, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6381306648254395, "reward_total_composite_std": 0.44584599137306213, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 89.0} {"timestamp_utc": "2026-04-11T19:32:27Z", "mode": "train", "global_step": 90, "epoch": 0.003475440222428174, "loss": 0.0647, "grad_norm": 12.855581283569336, "learning_rate": 9.730303030303031e-06, "num_tokens": 187088.0, "completions/mean_length": 66.5, "completions/min_length": 58.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.7145688533782959, "rewards/meter/std": 0.3503323197364807, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7145688533782959, "rewards/total_composite/std": 0.3503323197364807, "reward": 0.7145688533782959, "reward_std": 0.3503322899341583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18930701911449432, "sampling/sampling_logp_difference/max": 1.3135137557983398, "sampling/importance_sampling_ratio/min": 0.2688736319541931, "sampling/importance_sampling_ratio/mean": 1.0224878787994385, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0693052262067795, "clip_ratio/low_mean": 0.05007575824856758, "clip_ratio/low_min": 0.05007575824856758, "clip_ratio/high_mean": 0.1579482052475214, "clip_ratio/high_max": 0.1579482052475214, "clip_ratio/region_mean": 0.20802396349608898, "reward_total_mean": 0.7145688533782959, "reward_meter_mean": 0.7145688533782959, "reward_meter_std": 0.3503323197364807, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7145688533782959, "reward_total_composite_std": 0.3503323197364807, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 90.0} {"timestamp_utc": "2026-04-11T19:32:33Z", "mode": "train", "global_step": 91, "epoch": 0.0035140562248995983, "loss": 0.1231, "grad_norm": 6.48668909072876, "learning_rate": 9.727272727272728e-06, "num_tokens": 190233.0, "completions/mean_length": 192.125, "completions/min_length": 116.0, "completions/max_length": 231.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 192.125, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 231.0, "rewards/meter/mean": 0.9338054656982422, "rewards/meter/std": 0.08479016274213791, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.6788178086280823, "rewards/total_composite/std": 0.43503525853157043, "reward": 0.6788178086280823, "reward_std": 0.43503525853157043, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21253235638141632, "sampling/sampling_logp_difference/max": 1.493661880493164, "sampling/importance_sampling_ratio/min": 0.22454887628555298, "sampling/importance_sampling_ratio/mean": 1.0515210628509521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0097380578517914, "clip_ratio/low_mean": 0.06466159597039223, "clip_ratio/low_min": 0.06466159597039223, "clip_ratio/high_mean": 0.12906558997929096, "clip_ratio/high_max": 0.12906558997929096, "clip_ratio/region_mean": 0.1937271859496832, "reward_total_mean": 0.6788178086280823, "reward_meter_mean": 0.9338054656982422, "reward_meter_std": 0.08479016274213791, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.6788178086280823, "reward_total_composite_std": 0.43503525853157043, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 91.0} {"timestamp_utc": "2026-04-11T19:32:38Z", "mode": "train", "global_step": 92, "epoch": 0.0035526722273710224, "loss": -0.0808, "grad_norm": 15.7993745803833, "learning_rate": 9.724242424242426e-06, "num_tokens": 191874.0, "completions/mean_length": 54.125, "completions/min_length": 34.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.5675693154335022, "rewards/meter/std": 0.4166634678840637, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.532722532749176, "rewards/total_composite/std": 0.4542304575443268, "reward": 0.532722532749176, "reward_std": 0.4542304277420044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22817404568195343, "sampling/sampling_logp_difference/max": 1.7128546237945557, "sampling/importance_sampling_ratio/min": 0.18035022914409637, "sampling/importance_sampling_ratio/mean": 1.004146695137024, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.927680492401123, "clip_ratio/low_mean": 0.13832124322652817, "clip_ratio/low_min": 0.13832124322652817, "clip_ratio/high_mean": 0.11539113149046898, "clip_ratio/high_max": 0.11539113149046898, "clip_ratio/region_mean": 0.25371237471699715, "reward_total_mean": 0.532722532749176, "reward_meter_mean": 0.5675693154335022, "reward_meter_std": 0.4166634678840637, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.532722532749176, "reward_total_composite_std": 0.4542304575443268, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 92.0} {"timestamp_utc": "2026-04-11T19:32:46Z", "mode": "train", "global_step": 93, "epoch": 0.0035912882298424465, "loss": 0.0068, "grad_norm": 5.356238842010498, "learning_rate": 9.721212121212123e-06, "num_tokens": 196019.0, "completions/mean_length": 286.125, "completions/min_length": 259.0, "completions/max_length": 310.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 286.125, "completions/min_terminated_length": 259.0, "completions/max_terminated_length": 310.0, "rewards/meter/mean": 0.42547568678855896, "rewards/meter/std": 0.28940528631210327, "rewards/count_adherence/mean": 0.921875, "rewards/count_adherence/std": 0.06469365209341049, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.3027617931365967, "rewards/total_composite/std": 0.2622292935848236, "reward": 0.3027617931365967, "reward_std": 0.2622292637825012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19459845125675201, "sampling/sampling_logp_difference/max": 1.7290358543395996, "sampling/importance_sampling_ratio/min": 0.17745541036128998, "sampling/importance_sampling_ratio/mean": 1.0348670482635498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5551420152187347, "clip_ratio/low_mean": 0.10987469553947449, "clip_ratio/low_min": 0.10987469553947449, "clip_ratio/high_mean": 0.06774244643747807, "clip_ratio/high_max": 0.06774244643747807, "clip_ratio/region_mean": 0.17761714197695255, "reward_total_mean": 0.3027617931365967, "reward_meter_mean": 0.42547568678855896, "reward_meter_std": 0.28940528631210327, "reward_count_adherence_mean": 0.921875, "reward_count_adherence_std": 0.06469365209341049, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.3027617931365967, "reward_total_composite_std": 0.2622292935848236, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 93.0} {"timestamp_utc": "2026-04-11T19:32:51Z", "mode": "train", "global_step": 94, "epoch": 0.003629904232313871, "loss": -0.1159, "grad_norm": 11.193428993225098, "learning_rate": 9.718181818181818e-06, "num_tokens": 198136.0, "completions/mean_length": 98.625, "completions/min_length": 70.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.625, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.8099383115768433, "rewards/meter/std": 0.26826685667037964, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.6645107269287109, "rewards/total_composite/std": 0.4238337576389313, "reward": 0.6645107269287109, "reward_std": 0.4238337576389313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22230523824691772, "sampling/sampling_logp_difference/max": 1.9050703048706055, "sampling/importance_sampling_ratio/min": 0.1488121896982193, "sampling/importance_sampling_ratio/mean": 1.0548158884048462, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.608750492334366, "clip_ratio/low_mean": 0.049512987956404686, "clip_ratio/low_min": 0.049512987956404686, "clip_ratio/high_mean": 0.14526595920324326, "clip_ratio/high_max": 0.14526595920324326, "clip_ratio/region_mean": 0.19477894715964794, "reward_total_mean": 0.6645107269287109, "reward_meter_mean": 0.8099383115768433, "reward_meter_std": 0.26826685667037964, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.6645107269287109, "reward_total_composite_std": 0.4238337576389313, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 94.0} {"timestamp_utc": "2026-04-11T19:32:55Z", "mode": "train", "global_step": 95, "epoch": 0.003668520234785295, "loss": 0.1865, "grad_norm": 55.388946533203125, "learning_rate": 9.715151515151516e-06, "num_tokens": 199458.0, "completions/mean_length": 19.25, "completions/min_length": 18.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.25, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.9113181829452515, "rewards/meter/std": 0.23739001154899597, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9113181829452515, "rewards/total_composite/std": 0.23739001154899597, "reward": 0.9113181829452515, "reward_std": 0.23739001154899597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08319995552301407, "sampling/sampling_logp_difference/max": 1.0087409019470215, "sampling/importance_sampling_ratio/min": 0.36467787623405457, "sampling/importance_sampling_ratio/mean": 0.9945163726806641, "sampling/importance_sampling_ratio/max": 1.5570424795150757, "entropy": 0.30344432033598423, "clip_ratio/low_mean": 0.02777777798473835, "clip_ratio/low_min": 0.02777777798473835, "clip_ratio/high_mean": 0.047514619305729866, "clip_ratio/high_max": 0.047514619305729866, "clip_ratio/region_mean": 0.07529239729046822, "reward_total_mean": 0.9113181829452515, "reward_meter_mean": 0.9113181829452515, "reward_meter_std": 0.23739001154899597, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9113181829452515, "reward_total_composite_std": 0.23739001154899597, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 95.0} {"timestamp_utc": "2026-04-11T19:33:01Z", "mode": "train", "global_step": 96, "epoch": 0.0037071362372567192, "loss": 0.1775, "grad_norm": 11.280510902404785, "learning_rate": 9.712121212121213e-06, "num_tokens": 201486.0, "completions/mean_length": 94.5, "completions/min_length": 67.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.5, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.6954616904258728, "rewards/meter/std": 0.28914839029312134, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6954616904258728, "rewards/total_composite/std": 0.28914839029312134, "reward": 0.6954616904258728, "reward_std": 0.28914836049079895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2430281937122345, "sampling/sampling_logp_difference/max": 1.9791374206542969, "sampling/importance_sampling_ratio/min": 0.13818837702274323, "sampling/importance_sampling_ratio/mean": 1.037481665611267, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.304069072008133, "clip_ratio/low_mean": 0.07653119787573814, "clip_ratio/low_min": 0.07653119787573814, "clip_ratio/high_mean": 0.12667790986597538, "clip_ratio/high_max": 0.12667790986597538, "clip_ratio/region_mean": 0.20320910774171352, "reward_total_mean": 0.6954616904258728, "reward_meter_mean": 0.6954616904258728, "reward_meter_std": 0.28914839029312134, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6954616904258728, "reward_total_composite_std": 0.28914839029312134, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 96.0} {"timestamp_utc": "2026-04-11T19:33:06Z", "mode": "train", "global_step": 97, "epoch": 0.0037457522397281433, "loss": 0.046, "grad_norm": 11.798491477966309, "learning_rate": 9.70909090909091e-06, "num_tokens": 203468.0, "completions/mean_length": 77.75, "completions/min_length": 58.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.44233888387680054, "rewards/meter/std": 0.3354645371437073, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.44233888387680054, "rewards/total_composite/std": 0.3354645371437073, "reward": 0.44233888387680054, "reward_std": 0.3354645073413849, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23818424344062805, "sampling/sampling_logp_difference/max": 1.5784244537353516, "sampling/importance_sampling_ratio/min": 0.20629987120628357, "sampling/importance_sampling_ratio/mean": 1.0586459636688232, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.470811814069748, "clip_ratio/low_mean": 0.10330966301262379, "clip_ratio/low_min": 0.10330966301262379, "clip_ratio/high_mean": 0.11797327920794487, "clip_ratio/high_max": 0.11797327920794487, "clip_ratio/region_mean": 0.22128294222056866, "reward_total_mean": 0.44233888387680054, "reward_meter_mean": 0.44233888387680054, "reward_meter_std": 0.3354645371437073, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.44233888387680054, "reward_total_composite_std": 0.3354645371437073, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 97.0} {"timestamp_utc": "2026-04-11T19:33:11Z", "mode": "train", "global_step": 98, "epoch": 0.0037843682421995675, "loss": -0.0341, "grad_norm": 10.285151481628418, "learning_rate": 9.706060606060606e-06, "num_tokens": 205368.0, "completions/mean_length": 79.5, "completions/min_length": 59.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.5, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.458587646484375, "rewards/meter/std": 0.4029514789581299, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.364272803068161, "rewards/total_composite/std": 0.41265976428985596, "reward": 0.364272803068161, "reward_std": 0.41265976428985596, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22706757485866547, "sampling/sampling_logp_difference/max": 1.4814796447753906, "sampling/importance_sampling_ratio/min": 0.22730112075805664, "sampling/importance_sampling_ratio/mean": 1.0597635507583618, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.8207843005657196, "clip_ratio/low_mean": 0.12072680331766605, "clip_ratio/low_min": 0.12072680331766605, "clip_ratio/high_mean": 0.08083950355648994, "clip_ratio/high_max": 0.08083950355648994, "clip_ratio/region_mean": 0.201566306874156, "reward_total_mean": 0.364272803068161, "reward_meter_mean": 0.458587646484375, "reward_meter_std": 0.4029514789581299, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.364272803068161, "reward_total_composite_std": 0.41265976428985596, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 98.0} {"timestamp_utc": "2026-04-11T19:33:16Z", "mode": "train", "global_step": 99, "epoch": 0.0038229842446709916, "loss": 0.0857, "grad_norm": 11.669129371643066, "learning_rate": 9.703030303030305e-06, "num_tokens": 207481.0, "completions/mean_length": 89.125, "completions/min_length": 76.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.125, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.796177864074707, "rewards/meter/std": 0.30666306614875793, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.796177864074707, "rewards/total_composite/std": 0.30666306614875793, "reward": 0.796177864074707, "reward_std": 0.30666306614875793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16059812903404236, "sampling/sampling_logp_difference/max": 1.550947666168213, "sampling/importance_sampling_ratio/min": 0.21204693615436554, "sampling/importance_sampling_ratio/mean": 1.0176408290863037, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.236207775771618, "clip_ratio/low_mean": 0.062378629110753536, "clip_ratio/low_min": 0.062378629110753536, "clip_ratio/high_mean": 0.06744235754013062, "clip_ratio/high_max": 0.06744235754013062, "clip_ratio/region_mean": 0.12982098665088415, "reward_total_mean": 0.796177864074707, "reward_meter_mean": 0.796177864074707, "reward_meter_std": 0.30666306614875793, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.796177864074707, "reward_total_composite_std": 0.30666306614875793, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 99.0} {"timestamp_utc": "2026-04-11T19:33:20Z", "mode": "train", "global_step": 100, "epoch": 0.0038616002471424157, "loss": 0.0274, "grad_norm": 16.21261978149414, "learning_rate": 9.7e-06, "num_tokens": 209138.0, "completions/mean_length": 51.125, "completions/min_length": 43.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.125, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.4757336378097534, "rewards/meter/std": 0.39768749475479126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4757336378097534, "rewards/total_composite/std": 0.39768749475479126, "reward": 0.4757336378097534, "reward_std": 0.39768749475479126, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2088787704706192, "sampling/sampling_logp_difference/max": 1.368398666381836, "sampling/importance_sampling_ratio/min": 0.2545141875743866, "sampling/importance_sampling_ratio/mean": 1.024910569190979, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0875614881515503, "clip_ratio/low_mean": 0.0948890820145607, "clip_ratio/low_min": 0.0948890820145607, "clip_ratio/high_mean": 0.10258511640131474, "clip_ratio/high_max": 0.10258511640131474, "clip_ratio/region_mean": 0.19747419841587543, "reward_total_mean": 0.4757336378097534, "reward_meter_mean": 0.4757336378097534, "reward_meter_std": 0.39768749475479126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4757336378097534, "reward_total_composite_std": 0.39768749475479126, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 100.0} {"timestamp_utc": "2026-04-11T19:34:37Z", "mode": "eval", "global_step": 100, "epoch": 0.0038616002471424157, "eval_loss": NaN, "eval_runtime": 76.843, "eval_samples_per_second": 1.353, "eval_steps_per_second": 0.169, "eval_num_tokens": 209138.0, "eval_completions/mean_length": 189.1153846153846, "eval_completions/min_length": 39.15384615384615, "eval_completions/max_length": 412.9230769230769, "eval_completions/clipped_ratio": 0.038461538461538464, "eval_completions/mean_terminated_length": 176.4835181603065, "eval_completions/min_terminated_length": 39.15384615384615, "eval_completions/max_terminated_length": 374.53846153846155, "eval_rewards/meter/mean": 0.38446200467073, "eval_rewards/meter/std": 0.3474533214018895, "eval_rewards/count_adherence/mean": 0.9531642427811255, "eval_rewards/count_adherence/std": 0.06681363327571979, "eval_rewards/arabic_clean/mean": 0.9519230769230769, "eval_rewards/arabic_clean/std": 0.13598207097787124, "eval_rewards/total_composite/mean": 0.36039777329334843, "eval_rewards/total_composite/std": 0.34838862602527326, "eval_reward": 0.36039777329334843, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.1495220626776035, "eval_sampling/sampling_logp_difference/max": 1.3432146219106822, "eval_sampling/importance_sampling_ratio/min": 0.2675319703725668, "eval_sampling/importance_sampling_ratio/mean": 1.043038276525644, "eval_sampling/importance_sampling_ratio/max": 1.6659116469896758, "eval_entropy": 2.585876281444843, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.36039777329334843, "eval_reward_meter_mean": 0.38446200467073, "eval_reward_meter_std": 0.3474533214018895, "eval_reward_count_adherence_mean": 0.9531642427811255, "eval_reward_count_adherence_std": 0.06681363327571979, "eval_reward_arabic_clean_mean": 0.9519230769230769, "eval_reward_arabic_clean_std": 0.13598207097787124, "eval_reward_total_composite_mean": 0.36039777329334843, "eval_reward_total_composite_std": 0.34838862602527326, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 100.0} {"timestamp_utc": "2026-04-11T19:34:44Z", "mode": "train", "global_step": 101, "epoch": 0.0039002162496138398, "loss": -0.0129, "grad_norm": 13.529213905334473, "learning_rate": 9.696969696969698e-06, "num_tokens": 210953.0, "completions/mean_length": 64.875, "completions/min_length": 41.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.40993648767471313, "rewards/meter/std": 0.29322153329849243, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3698737621307373, "rewards/total_composite/std": 0.26750627160072327, "reward": 0.3698737621307373, "reward_std": 0.26750627160072327, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.203210711479187, "sampling/sampling_logp_difference/max": 1.2733283042907715, "sampling/importance_sampling_ratio/min": 0.2798984944820404, "sampling/importance_sampling_ratio/mean": 1.0172652006149292, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.11554679274559, "clip_ratio/low_mean": 0.10309341456741095, "clip_ratio/low_min": 0.10309341456741095, "clip_ratio/high_mean": 0.06029390636831522, "clip_ratio/high_max": 0.06029390636831522, "clip_ratio/region_mean": 0.16338732093572617, "reward_total_mean": 0.3698737621307373, "reward_meter_mean": 0.40993648767471313, "reward_meter_std": 0.29322153329849243, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3698737621307373, "reward_total_composite_std": 0.26750627160072327, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 101.0} {"timestamp_utc": "2026-04-11T19:34:49Z", "mode": "train", "global_step": 102, "epoch": 0.003938832252085264, "loss": -0.0229, "grad_norm": 9.461191177368164, "learning_rate": 9.693939393939395e-06, "num_tokens": 213210.0, "completions/mean_length": 109.125, "completions/min_length": 74.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.125, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.5020676851272583, "rewards/meter/std": 0.33497413992881775, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.3811372220516205, "rewards/total_composite/std": 0.3171204626560211, "reward": 0.3811372220516205, "reward_std": 0.3171204626560211, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2128397673368454, "sampling/sampling_logp_difference/max": 1.383610725402832, "sampling/importance_sampling_ratio/min": 0.2506718039512634, "sampling/importance_sampling_ratio/mean": 1.0544703006744385, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.91497141122818, "clip_ratio/low_mean": 0.09572393447160721, "clip_ratio/low_min": 0.09572393447160721, "clip_ratio/high_mean": 0.10505400598049164, "clip_ratio/high_max": 0.10505400598049164, "clip_ratio/region_mean": 0.20077794045209885, "reward_total_mean": 0.3811372220516205, "reward_meter_mean": 0.5020676851272583, "reward_meter_std": 0.33497413992881775, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.3811372220516205, "reward_total_composite_std": 0.3171204626560211, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 102.0} {"timestamp_utc": "2026-04-11T19:34:54Z", "mode": "train", "global_step": 103, "epoch": 0.003977448254556688, "loss": 0.0772, "grad_norm": 13.193228721618652, "learning_rate": 9.690909090909092e-06, "num_tokens": 215040.0, "completions/mean_length": 51.75, "completions/min_length": 36.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.48240411281585693, "rewards/meter/std": 0.4074939489364624, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.48240411281585693, "rewards/total_composite/std": 0.4074939489364624, "reward": 0.48240411281585693, "reward_std": 0.4074939489364624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.217100128531456, "sampling/sampling_logp_difference/max": 1.611802101135254, "sampling/importance_sampling_ratio/min": 0.19952771067619324, "sampling/importance_sampling_ratio/mean": 1.0494117736816406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3894866704940796, "clip_ratio/low_mean": 0.10796530079096556, "clip_ratio/low_min": 0.10796530079096556, "clip_ratio/high_mean": 0.08475394546985626, "clip_ratio/high_max": 0.08475394546985626, "clip_ratio/region_mean": 0.19271924626082182, "reward_total_mean": 0.48240411281585693, "reward_meter_mean": 0.48240411281585693, "reward_meter_std": 0.4074939489364624, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.48240411281585693, "reward_total_composite_std": 0.4074939489364624, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 103.0} {"timestamp_utc": "2026-04-11T19:34:58Z", "mode": "train", "global_step": 104, "epoch": 0.004016064257028112, "loss": -0.0086, "grad_norm": 14.126129150390625, "learning_rate": 9.687878787878788e-06, "num_tokens": 216684.0, "completions/mean_length": 46.5, "completions/min_length": 38.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6793533563613892, "rewards/meter/std": 0.37877920269966125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6793533563613892, "rewards/total_composite/std": 0.37877920269966125, "reward": 0.6793533563613892, "reward_std": 0.37877923250198364, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22054526209831238, "sampling/sampling_logp_difference/max": 1.4808216094970703, "sampling/importance_sampling_ratio/min": 0.22745074331760406, "sampling/importance_sampling_ratio/mean": 1.0407179594039917, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.75657719373703, "clip_ratio/low_mean": 0.07232538796961308, "clip_ratio/low_min": 0.07232538796961308, "clip_ratio/high_mean": 0.1372815314680338, "clip_ratio/high_max": 0.1372815314680338, "clip_ratio/region_mean": 0.20960691943764687, "reward_total_mean": 0.6793533563613892, "reward_meter_mean": 0.6793533563613892, "reward_meter_std": 0.37877920269966125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6793533563613892, "reward_total_composite_std": 0.37877920269966125, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 104.0} {"timestamp_utc": "2026-04-11T19:35:03Z", "mode": "train", "global_step": 105, "epoch": 0.004054680259499536, "loss": 0.0212, "grad_norm": 16.32758140563965, "learning_rate": 9.684848484848487e-06, "num_tokens": 218277.0, "completions/mean_length": 44.125, "completions/min_length": 40.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.4026336967945099, "rewards/meter/std": 0.35716599225997925, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4026336967945099, "rewards/total_composite/std": 0.35716599225997925, "reward": 0.4026336967945099, "reward_std": 0.35716599225997925, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16692768037319183, "sampling/sampling_logp_difference/max": 1.480295181274414, "sampling/importance_sampling_ratio/min": 0.2275705188512802, "sampling/importance_sampling_ratio/mean": 1.0534363985061646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7920314818620682, "clip_ratio/low_mean": 0.144812298938632, "clip_ratio/low_min": 0.144812298938632, "clip_ratio/high_mean": 0.06583333387970924, "clip_ratio/high_max": 0.06583333387970924, "clip_ratio/region_mean": 0.21064563281834126, "reward_total_mean": 0.4026336967945099, "reward_meter_mean": 0.4026336967945099, "reward_meter_std": 0.35716599225997925, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4026336967945099, "reward_total_composite_std": 0.35716599225997925, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 105.0} {"timestamp_utc": "2026-04-11T19:35:08Z", "mode": "train", "global_step": 106, "epoch": 0.004093296261970961, "loss": -0.0234, "grad_norm": 13.200583457946777, "learning_rate": 9.681818181818182e-06, "num_tokens": 220292.0, "completions/mean_length": 76.875, "completions/min_length": 54.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.875, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.3820000886917114, "rewards/meter/std": 0.3840147852897644, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.3593082129955292, "rewards/total_composite/std": 0.3966972827911377, "reward": 0.3593082129955292, "reward_std": 0.3966972529888153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24001426994800568, "sampling/sampling_logp_difference/max": 1.5144433975219727, "sampling/importance_sampling_ratio/min": 0.21993055939674377, "sampling/importance_sampling_ratio/mean": 1.0527663230895996, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.015815883874893, "clip_ratio/low_mean": 0.13967670314013958, "clip_ratio/low_min": 0.13967670314013958, "clip_ratio/high_mean": 0.10586144216358662, "clip_ratio/high_max": 0.10586144216358662, "clip_ratio/region_mean": 0.2455381453037262, "reward_total_mean": 0.3593082129955292, "reward_meter_mean": 0.3820000886917114, "reward_meter_std": 0.3840147852897644, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.3593082129955292, "reward_total_composite_std": 0.3966972827911377, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 106.0} {"timestamp_utc": "2026-04-11T19:35:16Z", "mode": "train", "global_step": 107, "epoch": 0.004131912264442385, "loss": 0.0184, "grad_norm": 5.8178300857543945, "learning_rate": 9.67878787878788e-06, "num_tokens": 223987.0, "completions/mean_length": 270.875, "completions/min_length": 164.0, "completions/max_length": 355.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 270.875, "completions/min_terminated_length": 164.0, "completions/max_terminated_length": 355.0, "rewards/meter/mean": 0.6609253883361816, "rewards/meter/std": 0.2803538143634796, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.06681530922651291, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5568721294403076, "rewards/total_composite/std": 0.3528820872306824, "reward": 0.5568721294403076, "reward_std": 0.3528820872306824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22177748382091522, "sampling/sampling_logp_difference/max": 1.5853900909423828, "sampling/importance_sampling_ratio/min": 0.2048678696155548, "sampling/importance_sampling_ratio/mean": 1.0432038307189941, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.4069304764270782, "clip_ratio/low_mean": 0.06378891877830029, "clip_ratio/low_min": 0.06378891877830029, "clip_ratio/high_mean": 0.1405453272163868, "clip_ratio/high_max": 0.1405453272163868, "clip_ratio/region_mean": 0.20433424599468708, "reward_total_mean": 0.5568721294403076, "reward_meter_mean": 0.6609253883361816, "reward_meter_std": 0.2803538143634796, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.06681530922651291, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5568721294403076, "reward_total_composite_std": 0.3528820872306824, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 107.0} {"timestamp_utc": "2026-04-11T19:35:20Z", "mode": "train", "global_step": 108, "epoch": 0.004170528266913809, "loss": 0.0163, "grad_norm": 16.24658203125, "learning_rate": 9.675757575757577e-06, "num_tokens": 225574.0, "completions/mean_length": 49.375, "completions/min_length": 35.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.375, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.4785667061805725, "rewards/meter/std": 0.40862172842025757, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.4669821858406067, "rewards/total_composite/std": 0.4222123324871063, "reward": 0.4669821858406067, "reward_std": 0.4222123324871063, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22205865383148193, "sampling/sampling_logp_difference/max": 1.374032974243164, "sampling/importance_sampling_ratio/min": 0.2530842125415802, "sampling/importance_sampling_ratio/mean": 1.0536181926727295, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.991265833377838, "clip_ratio/low_mean": 0.08765609562397003, "clip_ratio/low_min": 0.08765609562397003, "clip_ratio/high_mean": 0.10650285705924034, "clip_ratio/high_max": 0.10650285705924034, "clip_ratio/region_mean": 0.19415895268321037, "reward_total_mean": 0.4669821858406067, "reward_meter_mean": 0.4785667061805725, "reward_meter_std": 0.40862172842025757, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.4669821858406067, "reward_total_composite_std": 0.4222123324871063, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 108.0} {"timestamp_utc": "2026-04-11T19:35:26Z", "mode": "train", "global_step": 109, "epoch": 0.0042091442693852335, "loss": -0.0954, "grad_norm": 10.769991874694824, "learning_rate": 9.672727272727274e-06, "num_tokens": 227646.0, "completions/mean_length": 97.0, "completions/min_length": 66.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.525036096572876, "rewards/meter/std": 0.32685554027557373, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.525036096572876, "rewards/total_composite/std": 0.32685554027557373, "reward": 0.525036096572876, "reward_std": 0.32685554027557373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.230497807264328, "sampling/sampling_logp_difference/max": 1.2309355735778809, "sampling/importance_sampling_ratio/min": 0.292019248008728, "sampling/importance_sampling_ratio/mean": 1.024579405784607, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.290193408727646, "clip_ratio/low_mean": 0.07784751430153847, "clip_ratio/low_min": 0.07784751430153847, "clip_ratio/high_mean": 0.09419741854071617, "clip_ratio/high_max": 0.09419741854071617, "clip_ratio/region_mean": 0.17204493284225464, "reward_total_mean": 0.525036096572876, "reward_meter_mean": 0.525036096572876, "reward_meter_std": 0.32685554027557373, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.525036096572876, "reward_total_composite_std": 0.32685554027557373, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 109.0} {"timestamp_utc": "2026-04-11T19:35:30Z", "mode": "train", "global_step": 110, "epoch": 0.004247760271856658, "loss": -0.0629, "grad_norm": 13.498187065124512, "learning_rate": 9.66969696969697e-06, "num_tokens": 229401.0, "completions/mean_length": 62.375, "completions/min_length": 38.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.7822750210762024, "rewards/meter/std": 0.3426264226436615, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7822655439376831, "rewards/total_composite/std": 0.3426511585712433, "reward": 0.7822655439376831, "reward_std": 0.3426511585712433, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20653371512889862, "sampling/sampling_logp_difference/max": 1.3505010604858398, "sampling/importance_sampling_ratio/min": 0.3012003004550934, "sampling/importance_sampling_ratio/mean": 1.0487669706344604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1576623022556305, "clip_ratio/low_mean": 0.0400856789201498, "clip_ratio/low_min": 0.0400856789201498, "clip_ratio/high_mean": 0.11844915337860584, "clip_ratio/high_max": 0.11844915337860584, "clip_ratio/region_mean": 0.15853483229875565, "reward_total_mean": 0.7822655439376831, "reward_meter_mean": 0.7822750210762024, "reward_meter_std": 0.3426264226436615, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7822655439376831, "reward_total_composite_std": 0.3426511585712433, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 110.0} {"timestamp_utc": "2026-04-11T19:35:36Z", "mode": "train", "global_step": 111, "epoch": 0.004286376274328082, "loss": 0.0631, "grad_norm": 9.501466751098633, "learning_rate": 9.666666666666667e-06, "num_tokens": 231622.0, "completions/mean_length": 112.625, "completions/min_length": 72.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.625, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.6113415956497192, "rewards/meter/std": 0.39460012316703796, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5836721062660217, "rewards/total_composite/std": 0.3801315426826477, "reward": 0.5836721062660217, "reward_std": 0.3801315128803253, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1836644411087036, "sampling/sampling_logp_difference/max": 1.4742422103881836, "sampling/importance_sampling_ratio/min": 0.22895215451717377, "sampling/importance_sampling_ratio/mean": 1.0375341176986694, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.363381966948509, "clip_ratio/low_mean": 0.07732126116752625, "clip_ratio/low_min": 0.07732126116752625, "clip_ratio/high_mean": 0.10812411084771156, "clip_ratio/high_max": 0.10812411084771156, "clip_ratio/region_mean": 0.1854453720152378, "reward_total_mean": 0.5836721062660217, "reward_meter_mean": 0.6113415956497192, "reward_meter_std": 0.39460012316703796, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5836721062660217, "reward_total_composite_std": 0.3801315426826477, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 111.0} {"timestamp_utc": "2026-04-11T19:35:40Z", "mode": "train", "global_step": 112, "epoch": 0.004324992276799506, "loss": 0.1002, "grad_norm": 21.516904830932617, "learning_rate": 9.663636363636364e-06, "num_tokens": 233106.0, "completions/mean_length": 26.5, "completions/min_length": 18.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.5, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.30022066831588745, "rewards/meter/std": 0.3768041133880615, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.30022066831588745, "rewards/total_composite/std": 0.3768041133880615, "reward": 0.30022066831588745, "reward_std": 0.3768041133880615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20119836926460266, "sampling/sampling_logp_difference/max": 1.1516313552856445, "sampling/importance_sampling_ratio/min": 0.3161206543445587, "sampling/importance_sampling_ratio/mean": 1.041845679283142, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8556535243988037, "clip_ratio/low_mean": 0.10191993555054069, "clip_ratio/low_min": 0.10191993555054069, "clip_ratio/high_mean": 0.03674242552369833, "clip_ratio/high_max": 0.03674242552369833, "clip_ratio/region_mean": 0.13866236107423902, "reward_total_mean": 0.30022066831588745, "reward_meter_mean": 0.30022066831588745, "reward_meter_std": 0.3768041133880615, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.30022066831588745, "reward_total_composite_std": 0.3768041133880615, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 112.0} {"timestamp_utc": "2026-04-11T19:35:45Z", "mode": "train", "global_step": 113, "epoch": 0.00436360827927093, "loss": -0.0059, "grad_norm": 13.721908569335938, "learning_rate": 9.660606060606061e-06, "num_tokens": 234710.0, "completions/mean_length": 54.5, "completions/min_length": 38.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.543341875076294, "rewards/meter/std": 0.3554902970790863, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.543341875076294, "rewards/total_composite/std": 0.3554902970790863, "reward": 0.543341875076294, "reward_std": 0.3554903268814087, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22269107401371002, "sampling/sampling_logp_difference/max": 1.115030288696289, "sampling/importance_sampling_ratio/min": 0.3279053270816803, "sampling/importance_sampling_ratio/mean": 1.0506563186645508, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.915360987186432, "clip_ratio/low_mean": 0.0982854887843132, "clip_ratio/low_min": 0.0982854887843132, "clip_ratio/high_mean": 0.12023117486387491, "clip_ratio/high_max": 0.12023117486387491, "clip_ratio/region_mean": 0.21851666364818811, "reward_total_mean": 0.543341875076294, "reward_meter_mean": 0.543341875076294, "reward_meter_std": 0.3554902970790863, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.543341875076294, "reward_total_composite_std": 0.3554902970790863, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 113.0} {"timestamp_utc": "2026-04-11T19:35:50Z", "mode": "train", "global_step": 114, "epoch": 0.004402224281742354, "loss": -0.0029, "grad_norm": 10.23147964477539, "learning_rate": 9.657575757575758e-06, "num_tokens": 236805.0, "completions/mean_length": 85.875, "completions/min_length": 65.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.21813370287418365, "rewards/meter/std": 0.3496148884296417, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.21646319329738617, "rewards/total_composite/std": 0.35061758756637573, "reward": 0.21646319329738617, "reward_std": 0.35061758756637573, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19407296180725098, "sampling/sampling_logp_difference/max": 1.217595100402832, "sampling/importance_sampling_ratio/min": 0.295941025018692, "sampling/importance_sampling_ratio/mean": 1.0350022315979004, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.166059359908104, "clip_ratio/low_mean": 0.11057115998119116, "clip_ratio/low_min": 0.11057115998119116, "clip_ratio/high_mean": 0.0366877019405365, "clip_ratio/high_max": 0.0366877019405365, "clip_ratio/region_mean": 0.14725886192172766, "reward_total_mean": 0.21646319329738617, "reward_meter_mean": 0.21813370287418365, "reward_meter_std": 0.3496148884296417, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.21646319329738617, "reward_total_composite_std": 0.35061758756637573, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 114.0} {"timestamp_utc": "2026-04-11T19:35:55Z", "mode": "train", "global_step": 115, "epoch": 0.004440840284213778, "loss": 0.0735, "grad_norm": 9.909034729003906, "learning_rate": 9.654545454545456e-06, "num_tokens": 239099.0, "completions/mean_length": 117.75, "completions/min_length": 98.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.75, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.4193575978279114, "rewards/meter/std": 0.3706149458885193, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4193575978279114, "rewards/total_composite/std": 0.3706149458885193, "reward": 0.4193575978279114, "reward_std": 0.3706149160861969, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21766838431358337, "sampling/sampling_logp_difference/max": 1.9731688499450684, "sampling/importance_sampling_ratio/min": 0.13901562988758087, "sampling/importance_sampling_ratio/mean": 1.024393916130066, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1785334795713425, "clip_ratio/low_mean": 0.11949973180890083, "clip_ratio/low_min": 0.11949973180890083, "clip_ratio/high_mean": 0.08701479807496071, "clip_ratio/high_max": 0.08701479807496071, "clip_ratio/region_mean": 0.20651452988386154, "reward_total_mean": 0.4193575978279114, "reward_meter_mean": 0.4193575978279114, "reward_meter_std": 0.3706149458885193, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4193575978279114, "reward_total_composite_std": 0.3706149458885193, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 115.0} {"timestamp_utc": "2026-04-11T19:36:00Z", "mode": "train", "global_step": 116, "epoch": 0.004479456286685202, "loss": 0.0996, "grad_norm": 13.37387752532959, "learning_rate": 9.651515151515153e-06, "num_tokens": 240788.0, "completions/mean_length": 56.125, "completions/min_length": 38.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.6912834644317627, "rewards/meter/std": 0.39879509806632996, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6912834644317627, "rewards/total_composite/std": 0.39879509806632996, "reward": 0.6912834644317627, "reward_std": 0.39879506826400757, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18541668355464935, "sampling/sampling_logp_difference/max": 1.529659628868103, "sampling/importance_sampling_ratio/min": 0.216609388589859, "sampling/importance_sampling_ratio/mean": 1.0366685390472412, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4604494273662567, "clip_ratio/low_mean": 0.05248366016894579, "clip_ratio/low_min": 0.05248366016894579, "clip_ratio/high_mean": 0.15062034130096436, "clip_ratio/high_max": 0.15062034130096436, "clip_ratio/region_mean": 0.20310400146991014, "reward_total_mean": 0.6912834644317627, "reward_meter_mean": 0.6912834644317627, "reward_meter_std": 0.39879509806632996, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6912834644317627, "reward_total_composite_std": 0.39879509806632996, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 116.0} {"timestamp_utc": "2026-04-11T19:36:04Z", "mode": "train", "global_step": 117, "epoch": 0.004518072289156626, "loss": 0.1027, "grad_norm": 22.580102920532227, "learning_rate": 9.648484848484849e-06, "num_tokens": 242291.0, "completions/mean_length": 24.875, "completions/min_length": 16.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.875, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.4409791827201843, "rewards/meter/std": 0.46491894125938416, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.4115869998931885, "rewards/total_composite/std": 0.48671314120292664, "reward": 0.4115869998931885, "reward_std": 0.48671311140060425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2450607866048813, "sampling/sampling_logp_difference/max": 1.5653746128082275, "sampling/importance_sampling_ratio/min": 0.20900969207286835, "sampling/importance_sampling_ratio/mean": 1.0259357690811157, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.409518927335739, "clip_ratio/low_mean": 0.15885628014802933, "clip_ratio/low_min": 0.15885628014802933, "clip_ratio/high_mean": 0.0818505771458149, "clip_ratio/high_max": 0.0818505771458149, "clip_ratio/region_mean": 0.24070685729384422, "reward_total_mean": 0.4115869998931885, "reward_meter_mean": 0.4409791827201843, "reward_meter_std": 0.46491894125938416, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.4115869998931885, "reward_total_composite_std": 0.48671314120292664, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 117.0} {"timestamp_utc": "2026-04-11T19:36:11Z", "mode": "train", "global_step": 118, "epoch": 0.00455668829162805, "loss": 0.0803, "grad_norm": 8.194822311401367, "learning_rate": 9.645454545454548e-06, "num_tokens": 244935.0, "completions/mean_length": 159.5, "completions/min_length": 124.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.5, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.6182658672332764, "rewards/meter/std": 0.35096442699432373, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6134665012359619, "rewards/total_composite/std": 0.3578221797943115, "reward": 0.6134665012359619, "reward_std": 0.3578221797943115, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20374369621276855, "sampling/sampling_logp_difference/max": 1.8341115713119507, "sampling/importance_sampling_ratio/min": 0.1597553789615631, "sampling/importance_sampling_ratio/mean": 1.02511465549469, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5699357092380524, "clip_ratio/low_mean": 0.0561376241967082, "clip_ratio/low_min": 0.0561376241967082, "clip_ratio/high_mean": 0.14602509513497353, "clip_ratio/high_max": 0.14602509513497353, "clip_ratio/region_mean": 0.20216271933168173, "reward_total_mean": 0.6134665012359619, "reward_meter_mean": 0.6182658672332764, "reward_meter_std": 0.35096442699432373, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6134665012359619, "reward_total_composite_std": 0.3578221797943115, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 118.0} {"timestamp_utc": "2026-04-11T19:36:15Z", "mode": "train", "global_step": 119, "epoch": 0.0045953042940994745, "loss": -0.0893, "grad_norm": 15.294441223144531, "learning_rate": 9.642424242424243e-06, "num_tokens": 246392.0, "completions/mean_length": 29.125, "completions/min_length": 20.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.125, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9771455526351929, "rewards/meter/std": 0.02793486975133419, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8526638150215149, "rewards/total_composite/std": 0.3455761969089508, "reward": 0.8526638150215149, "reward_std": 0.3455761969089508, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21847626566886902, "sampling/sampling_logp_difference/max": 2.0008230209350586, "sampling/importance_sampling_ratio/min": 0.13522395491600037, "sampling/importance_sampling_ratio/mean": 1.0416526794433594, "sampling/importance_sampling_ratio/max": 1.9315763711929321, "entropy": 2.4730520844459534, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/high_mean": 0.18722888734191656, "clip_ratio/high_max": 0.18722888734191656, "clip_ratio/region_mean": 0.19972888752818108, "reward_total_mean": 0.8526638150215149, "reward_meter_mean": 0.9771455526351929, "reward_meter_std": 0.02793486975133419, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8526638150215149, "reward_total_composite_std": 0.3455761969089508, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 119.0} {"timestamp_utc": "2026-04-11T19:36:20Z", "mode": "train", "global_step": 120, "epoch": 0.004633920296570899, "loss": 0.1424, "grad_norm": 13.367688179016113, "learning_rate": 9.63939393939394e-06, "num_tokens": 248367.0, "completions/mean_length": 72.875, "completions/min_length": 50.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.875, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.3219662308692932, "rewards/meter/std": 0.37545910477638245, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.30540409684181213, "rewards/total_composite/std": 0.3886590600013733, "reward": 0.30540409684181213, "reward_std": 0.3886590600013733, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23424479365348816, "sampling/sampling_logp_difference/max": 1.8647069931030273, "sampling/importance_sampling_ratio/min": 0.1549416035413742, "sampling/importance_sampling_ratio/mean": 1.0693588256835938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9914455711841583, "clip_ratio/low_mean": 0.16581876948475838, "clip_ratio/low_min": 0.16581876948475838, "clip_ratio/high_mean": 0.0674145333468914, "clip_ratio/high_max": 0.0674145333468914, "clip_ratio/region_mean": 0.23323330283164978, "reward_total_mean": 0.30540409684181213, "reward_meter_mean": 0.3219662308692932, "reward_meter_std": 0.37545910477638245, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.30540409684181213, "reward_total_composite_std": 0.3886590600013733, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 120.0} {"timestamp_utc": "2026-04-11T19:36:26Z", "mode": "train", "global_step": 121, "epoch": 0.004672536299042323, "loss": 0.0533, "grad_norm": 8.033286094665527, "learning_rate": 9.636363636363638e-06, "num_tokens": 251459.0, "completions/mean_length": 164.5, "completions/min_length": 157.0, "completions/max_length": 174.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 164.5, "completions/min_terminated_length": 157.0, "completions/max_terminated_length": 174.0, "rewards/meter/mean": 0.7737554907798767, "rewards/meter/std": 0.31233546137809753, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7737554907798767, "rewards/total_composite/std": 0.31233546137809753, "reward": 0.7737554907798767, "reward_std": 0.31233546137809753, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17217814922332764, "sampling/sampling_logp_difference/max": 1.9242980480194092, "sampling/importance_sampling_ratio/min": 0.1459781974554062, "sampling/importance_sampling_ratio/mean": 1.0243695974349976, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8652900010347366, "clip_ratio/low_mean": 0.07312539592385292, "clip_ratio/low_min": 0.07312539592385292, "clip_ratio/high_mean": 0.11291690729558468, "clip_ratio/high_max": 0.11291690729558468, "clip_ratio/region_mean": 0.1860423032194376, "reward_total_mean": 0.7737554907798767, "reward_meter_mean": 0.7737554907798767, "reward_meter_std": 0.31233546137809753, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7737554907798767, "reward_total_composite_std": 0.31233546137809753, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 121.0} {"timestamp_utc": "2026-04-11T19:36:30Z", "mode": "train", "global_step": 122, "epoch": 0.004711152301513748, "loss": 0.0725, "grad_norm": 21.118961334228516, "learning_rate": 9.633333333333335e-06, "num_tokens": 252974.0, "completions/mean_length": 29.375, "completions/min_length": 25.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.4754229187965393, "rewards/meter/std": 0.4443075954914093, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4754229187965393, "rewards/total_composite/std": 0.4443075954914093, "reward": 0.4754229187965393, "reward_std": 0.4443075358867645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2008654922246933, "sampling/sampling_logp_difference/max": 1.3499740362167358, "sampling/importance_sampling_ratio/min": 0.25924697518348694, "sampling/importance_sampling_ratio/mean": 1.0281100273132324, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7652578055858612, "clip_ratio/low_mean": 0.12684562429785728, "clip_ratio/low_min": 0.12684562429785728, "clip_ratio/high_mean": 0.10151049681007862, "clip_ratio/high_max": 0.10151049681007862, "clip_ratio/region_mean": 0.2283561211079359, "reward_total_mean": 0.4754229187965393, "reward_meter_mean": 0.4754229187965393, "reward_meter_std": 0.4443075954914093, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4754229187965393, "reward_total_composite_std": 0.4443075954914093, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 122.0} {"timestamp_utc": "2026-04-11T19:36:35Z", "mode": "train", "global_step": 123, "epoch": 0.004749768303985172, "loss": -0.0038, "grad_norm": 10.26867961883545, "learning_rate": 9.63030303030303e-06, "num_tokens": 255275.0, "completions/mean_length": 106.625, "completions/min_length": 86.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.625, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9178913831710815, "rewards/meter/std": 0.17877134680747986, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9178913831710815, "rewards/total_composite/std": 0.17877134680747986, "reward": 0.9178913831710815, "reward_std": 0.17877133190631866, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20389947295188904, "sampling/sampling_logp_difference/max": 1.5918359756469727, "sampling/importance_sampling_ratio/min": 0.2035515457391739, "sampling/importance_sampling_ratio/mean": 1.0357023477554321, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5758557319641113, "clip_ratio/low_mean": 0.033740474842488766, "clip_ratio/low_min": 0.033740474842488766, "clip_ratio/high_mean": 0.14564990997314453, "clip_ratio/high_max": 0.14564990997314453, "clip_ratio/region_mean": 0.1793903848156333, "reward_total_mean": 0.9178913831710815, "reward_meter_mean": 0.9178913831710815, "reward_meter_std": 0.17877134680747986, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9178913831710815, "reward_total_composite_std": 0.17877134680747986, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 123.0} {"timestamp_utc": "2026-04-11T19:36:45Z", "mode": "train", "global_step": 124, "epoch": 0.004788384306456596, "loss": 0.0314, "grad_norm": 4.5853095054626465, "learning_rate": 9.627272727272728e-06, "num_tokens": 259818.0, "completions/mean_length": 340.875, "completions/min_length": 158.0, "completions/max_length": 474.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 340.875, "completions/min_terminated_length": 158.0, "completions/max_terminated_length": 474.0, "rewards/meter/mean": 0.5577775239944458, "rewards/meter/std": 0.29362478852272034, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.1679842174053192, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.3518419861793518, "rewards/total_composite/std": 0.3233281672000885, "reward": 0.3518419861793518, "reward_std": 0.3233281970024109, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2014904022216797, "sampling/sampling_logp_difference/max": 1.6749906539916992, "sampling/importance_sampling_ratio/min": 0.18730992078781128, "sampling/importance_sampling_ratio/mean": 1.033653974533081, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0468600690364838, "clip_ratio/low_mean": 0.09205890912562609, "clip_ratio/low_min": 0.09205890912562609, "clip_ratio/high_mean": 0.07448212616145611, "clip_ratio/high_max": 0.07448212616145611, "clip_ratio/region_mean": 0.1665410352870822, "reward_total_mean": 0.3518419861793518, "reward_meter_mean": 0.5577775239944458, "reward_meter_std": 0.29362478852272034, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.1679842174053192, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.3518419861793518, "reward_total_composite_std": 0.3233281672000885, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 124.0} {"timestamp_utc": "2026-04-11T19:36:49Z", "mode": "train", "global_step": 125, "epoch": 0.00482700030892802, "loss": 0.1625, "grad_norm": 17.506608963012695, "learning_rate": 9.624242424242425e-06, "num_tokens": 261528.0, "completions/mean_length": 54.75, "completions/min_length": 41.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.7977160215377808, "rewards/meter/std": 0.3083810806274414, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7977160215377808, "rewards/total_composite/std": 0.3083810806274414, "reward": 0.7977160215377808, "reward_std": 0.308381050825119, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22570040822029114, "sampling/sampling_logp_difference/max": 1.3436055183410645, "sampling/importance_sampling_ratio/min": 0.2609032988548279, "sampling/importance_sampling_ratio/mean": 1.0523124933242798, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.44717738032341, "clip_ratio/low_mean": 0.03829259052872658, "clip_ratio/low_min": 0.03829259052872658, "clip_ratio/high_mean": 0.15175942331552505, "clip_ratio/high_max": 0.15175942331552505, "clip_ratio/region_mean": 0.19005201384425163, "reward_total_mean": 0.7977160215377808, "reward_meter_mean": 0.7977160215377808, "reward_meter_std": 0.3083810806274414, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7977160215377808, "reward_total_composite_std": 0.3083810806274414, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 125.0} {"timestamp_utc": "2026-04-11T19:36:55Z", "mode": "train", "global_step": 126, "epoch": 0.004865616311399444, "loss": 0.2206, "grad_norm": 13.348600387573242, "learning_rate": 9.621212121212122e-06, "num_tokens": 263761.0, "completions/mean_length": 108.125, "completions/min_length": 85.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.125, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.5293493866920471, "rewards/meter/std": 0.396657794713974, "rewards/count_adherence/mean": 0.9285714626312256, "rewards/count_adherence/std": 0.07636035233736038, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5052797198295593, "rewards/total_composite/std": 0.39642462134361267, "reward": 0.5052797198295593, "reward_std": 0.39642462134361267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18913644552230835, "sampling/sampling_logp_difference/max": 2.8120975494384766, "sampling/importance_sampling_ratio/min": 0.06007884442806244, "sampling/importance_sampling_ratio/mean": 1.02097487449646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7009220086038113, "clip_ratio/low_mean": 0.10175232589244843, "clip_ratio/low_min": 0.10175232589244843, "clip_ratio/high_mean": 0.026411979109980166, "clip_ratio/high_max": 0.026411979109980166, "clip_ratio/region_mean": 0.1281643050024286, "reward_total_mean": 0.5052797198295593, "reward_meter_mean": 0.5293493866920471, "reward_meter_std": 0.396657794713974, "reward_count_adherence_mean": 0.9285714626312256, "reward_count_adherence_std": 0.07636035233736038, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5052797198295593, "reward_total_composite_std": 0.39642462134361267, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 126.0} {"timestamp_utc": "2026-04-11T19:36:59Z", "mode": "train", "global_step": 127, "epoch": 0.004904232313870868, "loss": 0.0506, "grad_norm": 14.61343002319336, "learning_rate": 9.61818181818182e-06, "num_tokens": 265393.0, "completions/mean_length": 48.0, "completions/min_length": 35.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.3801646828651428, "rewards/meter/std": 0.4593394994735718, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3801646828651428, "rewards/total_composite/std": 0.4593394994735718, "reward": 0.3801646828651428, "reward_std": 0.4593394994735718, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19850540161132812, "sampling/sampling_logp_difference/max": 1.1754345893859863, "sampling/importance_sampling_ratio/min": 0.3086847960948944, "sampling/importance_sampling_ratio/mean": 1.0501198768615723, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.404453784227371, "clip_ratio/low_mean": 0.11877263803035021, "clip_ratio/low_min": 0.11877263803035021, "clip_ratio/high_mean": 0.08253205195069313, "clip_ratio/high_max": 0.08253205195069313, "clip_ratio/region_mean": 0.20130468998104334, "reward_total_mean": 0.3801646828651428, "reward_meter_mean": 0.3801646828651428, "reward_meter_std": 0.4593394994735718, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3801646828651428, "reward_total_composite_std": 0.4593394994735718, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 127.0} {"timestamp_utc": "2026-04-11T19:37:04Z", "mode": "train", "global_step": 128, "epoch": 0.004942848316342292, "loss": 0.035, "grad_norm": 14.53402042388916, "learning_rate": 9.615151515151517e-06, "num_tokens": 267018.0, "completions/mean_length": 55.125, "completions/min_length": 45.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.125, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7447742223739624, "rewards/meter/std": 0.3583148717880249, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7447742223739624, "rewards/total_composite/std": 0.3583148717880249, "reward": 0.7447742223739624, "reward_std": 0.3583148717880249, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20606549084186554, "sampling/sampling_logp_difference/max": 1.2611989974975586, "sampling/importance_sampling_ratio/min": 0.2833141088485718, "sampling/importance_sampling_ratio/mean": 1.042881965637207, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.133269816637039, "clip_ratio/low_mean": 0.04620295576751232, "clip_ratio/low_min": 0.04620295576751232, "clip_ratio/high_mean": 0.14333923161029816, "clip_ratio/high_max": 0.14333923161029816, "clip_ratio/region_mean": 0.18954218737781048, "reward_total_mean": 0.7447742223739624, "reward_meter_mean": 0.7447742223739624, "reward_meter_std": 0.3583148717880249, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7447742223739624, "reward_total_composite_std": 0.3583148717880249, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 128.0} {"timestamp_utc": "2026-04-11T19:37:08Z", "mode": "train", "global_step": 129, "epoch": 0.004981464318813716, "loss": 0.2607, "grad_norm": 23.683666229248047, "learning_rate": 9.612121212121212e-06, "num_tokens": 268338.0, "completions/mean_length": 24.0, "completions/min_length": 15.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.0, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.6895738840103149, "rewards/meter/std": 0.3600609004497528, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6895738840103149, "rewards/total_composite/std": 0.3600609004497528, "reward": 0.6895738840103149, "reward_std": 0.3600608706474304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21219754219055176, "sampling/sampling_logp_difference/max": 1.3378047943115234, "sampling/importance_sampling_ratio/min": 0.2624211013317108, "sampling/importance_sampling_ratio/mean": 1.022581696510315, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7079626321792603, "clip_ratio/low_mean": 0.05078725144267082, "clip_ratio/low_min": 0.05078725144267082, "clip_ratio/high_mean": 0.10729072242975235, "clip_ratio/high_max": 0.10729072242975235, "clip_ratio/region_mean": 0.15807797387242317, "reward_total_mean": 0.6895738840103149, "reward_meter_mean": 0.6895738840103149, "reward_meter_std": 0.3600609004497528, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6895738840103149, "reward_total_composite_std": 0.3600609004497528, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 129.0} {"timestamp_utc": "2026-04-11T19:37:13Z", "mode": "train", "global_step": 130, "epoch": 0.0050200803212851405, "loss": 0.0564, "grad_norm": 13.415695190429688, "learning_rate": 9.60909090909091e-06, "num_tokens": 270058.0, "completions/mean_length": 53.0, "completions/min_length": 43.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.0, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.2599804103374481, "rewards/meter/std": 0.33920830488204956, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.2599804103374481, "rewards/total_composite/std": 0.33920830488204956, "reward": 0.2599804103374481, "reward_std": 0.33920830488204956, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2094995379447937, "sampling/sampling_logp_difference/max": 1.2830052375793457, "sampling/importance_sampling_ratio/min": 0.2772029638290405, "sampling/importance_sampling_ratio/mean": 1.0430470705032349, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.128389284014702, "clip_ratio/low_mean": 0.10636867955327034, "clip_ratio/low_min": 0.10636867955327034, "clip_ratio/high_mean": 0.07609842158854008, "clip_ratio/high_max": 0.07609842158854008, "clip_ratio/region_mean": 0.18246710114181042, "reward_total_mean": 0.2599804103374481, "reward_meter_mean": 0.2599804103374481, "reward_meter_std": 0.33920830488204956, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.2599804103374481, "reward_total_composite_std": 0.33920830488204956, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 130.0} {"timestamp_utc": "2026-04-11T19:37:17Z", "mode": "train", "global_step": 131, "epoch": 0.005058696323756565, "loss": 0.0505, "grad_norm": 17.473791122436523, "learning_rate": 9.606060606060607e-06, "num_tokens": 271776.0, "completions/mean_length": 54.75, "completions/min_length": 36.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.6479246616363525, "rewards/meter/std": 0.4171454906463623, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6479246616363525, "rewards/total_composite/std": 0.4171454906463623, "reward": 0.6479246616363525, "reward_std": 0.41714543104171753, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21375833451747894, "sampling/sampling_logp_difference/max": 1.422128677368164, "sampling/importance_sampling_ratio/min": 0.2412000447511673, "sampling/importance_sampling_ratio/mean": 1.0370509624481201, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6511287838220596, "clip_ratio/low_mean": 0.07387228682637215, "clip_ratio/low_min": 0.07387228682637215, "clip_ratio/high_mean": 0.1310401326045394, "clip_ratio/high_max": 0.1310401326045394, "clip_ratio/region_mean": 0.20491241943091154, "reward_total_mean": 0.6479246616363525, "reward_meter_mean": 0.6479246616363525, "reward_meter_std": 0.4171454906463623, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6479246616363525, "reward_total_composite_std": 0.4171454906463623, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 131.0} {"timestamp_utc": "2026-04-11T19:37:22Z", "mode": "train", "global_step": 132, "epoch": 0.005097312326227989, "loss": 0.0909, "grad_norm": 15.81747817993164, "learning_rate": 9.603030303030304e-06, "num_tokens": 273490.0, "completions/mean_length": 45.25, "completions/min_length": 36.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.2023978978395462, "rewards/meter/std": 0.27117809653282166, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.2023978978395462, "rewards/total_composite/std": 0.27117809653282166, "reward": 0.2023978978395462, "reward_std": 0.27117806673049927, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2216556966304779, "sampling/sampling_logp_difference/max": 1.3197402954101562, "sampling/importance_sampling_ratio/min": 0.2672046720981598, "sampling/importance_sampling_ratio/mean": 1.0494537353515625, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1410828977823257, "clip_ratio/low_mean": 0.09321646392345428, "clip_ratio/low_min": 0.09321646392345428, "clip_ratio/high_mean": 0.07182539906352758, "clip_ratio/high_max": 0.07182539906352758, "clip_ratio/region_mean": 0.16504186298698187, "reward_total_mean": 0.2023978978395462, "reward_meter_mean": 0.2023978978395462, "reward_meter_std": 0.27117809653282166, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.2023978978395462, "reward_total_composite_std": 0.27117809653282166, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 132.0} {"timestamp_utc": "2026-04-11T19:37:26Z", "mode": "train", "global_step": 133, "epoch": 0.005135928328699413, "loss": -0.0793, "grad_norm": 15.763381004333496, "learning_rate": 9.600000000000001e-06, "num_tokens": 275138.0, "completions/mean_length": 49.0, "completions/min_length": 37.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.5731572508811951, "rewards/meter/std": 0.42340850830078125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5731572508811951, "rewards/total_composite/std": 0.42340850830078125, "reward": 0.5731572508811951, "reward_std": 0.42340850830078125, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21350626647472382, "sampling/sampling_logp_difference/max": 1.368790626525879, "sampling/importance_sampling_ratio/min": 0.254414439201355, "sampling/importance_sampling_ratio/mean": 1.0395585298538208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9459713995456696, "clip_ratio/low_mean": 0.10132601391524076, "clip_ratio/low_min": 0.10132601391524076, "clip_ratio/high_mean": 0.10300109349191189, "clip_ratio/high_max": 0.10300109349191189, "clip_ratio/region_mean": 0.20432710740715265, "reward_total_mean": 0.5731572508811951, "reward_meter_mean": 0.5731572508811951, "reward_meter_std": 0.42340850830078125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5731572508811951, "reward_total_composite_std": 0.42340850830078125, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 133.0} {"timestamp_utc": "2026-04-11T19:37:31Z", "mode": "train", "global_step": 134, "epoch": 0.005174544331170837, "loss": -0.0251, "grad_norm": 15.412650108337402, "learning_rate": 9.596969696969699e-06, "num_tokens": 276875.0, "completions/mean_length": 57.125, "completions/min_length": 41.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7134405970573425, "rewards/meter/std": 0.36172330379486084, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7134405970573425, "rewards/total_composite/std": 0.36172330379486084, "reward": 0.7134405970573425, "reward_std": 0.36172330379486084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20017951726913452, "sampling/sampling_logp_difference/max": 1.7018890380859375, "sampling/importance_sampling_ratio/min": 0.18233875930309296, "sampling/importance_sampling_ratio/mean": 1.0319643020629883, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2369899600744247, "clip_ratio/low_mean": 0.05536198429763317, "clip_ratio/low_min": 0.05536198429763317, "clip_ratio/high_mean": 0.1634557507932186, "clip_ratio/high_max": 0.1634557507932186, "clip_ratio/region_mean": 0.21881773509085178, "reward_total_mean": 0.7134405970573425, "reward_meter_mean": 0.7134405970573425, "reward_meter_std": 0.36172330379486084, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7134405970573425, "reward_total_composite_std": 0.36172330379486084, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 134.0} {"timestamp_utc": "2026-04-11T19:37:38Z", "mode": "train", "global_step": 135, "epoch": 0.005213160333642261, "loss": 0.009, "grad_norm": 6.406003952026367, "learning_rate": 9.593939393939394e-06, "num_tokens": 279842.0, "completions/mean_length": 194.875, "completions/min_length": 125.0, "completions/max_length": 281.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 194.875, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 281.0, "rewards/meter/mean": 0.3760119676589966, "rewards/meter/std": 0.4099160134792328, "rewards/count_adherence/mean": 0.9464285969734192, "rewards/count_adherence/std": 0.07393559068441391, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.3583834171295166, "rewards/total_composite/std": 0.4076659679412842, "reward": 0.3583834171295166, "reward_std": 0.4076659679412842, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1984342634677887, "sampling/sampling_logp_difference/max": 1.539616584777832, "sampling/importance_sampling_ratio/min": 0.21446330845355988, "sampling/importance_sampling_ratio/mean": 1.0401254892349243, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0369058549404144, "clip_ratio/low_mean": 0.09396191779524088, "clip_ratio/low_min": 0.09396191779524088, "clip_ratio/high_mean": 0.07667344622313976, "clip_ratio/high_max": 0.07667344622313976, "clip_ratio/region_mean": 0.17063536401838064, "reward_total_mean": 0.3583834171295166, "reward_meter_mean": 0.3760119676589966, "reward_meter_std": 0.4099160134792328, "reward_count_adherence_mean": 0.9464285969734192, "reward_count_adherence_std": 0.07393559068441391, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.3583834171295166, "reward_total_composite_std": 0.4076659679412842, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 135.0} {"timestamp_utc": "2026-04-11T19:37:42Z", "mode": "train", "global_step": 136, "epoch": 0.005251776336113685, "loss": -0.0186, "grad_norm": 13.780417442321777, "learning_rate": 9.590909090909091e-06, "num_tokens": 281484.0, "completions/mean_length": 51.25, "completions/min_length": 39.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.25, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.6177759170532227, "rewards/meter/std": 0.46374326944351196, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6177759170532227, "rewards/total_composite/std": 0.46374326944351196, "reward": 0.6177759170532227, "reward_std": 0.4637432098388672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.210972860455513, "sampling/sampling_logp_difference/max": 1.8416433334350586, "sampling/importance_sampling_ratio/min": 0.15855665504932404, "sampling/importance_sampling_ratio/mean": 1.0363706350326538, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2263315618038177, "clip_ratio/low_mean": 0.06984265986829996, "clip_ratio/low_min": 0.06984265986829996, "clip_ratio/high_mean": 0.11762434057891369, "clip_ratio/high_max": 0.11762434057891369, "clip_ratio/region_mean": 0.18746700044721365, "reward_total_mean": 0.6177759170532227, "reward_meter_mean": 0.6177759170532227, "reward_meter_std": 0.46374326944351196, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6177759170532227, "reward_total_composite_std": 0.46374326944351196, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 136.0} {"timestamp_utc": "2026-04-11T19:37:47Z", "mode": "train", "global_step": 137, "epoch": 0.005290392338585109, "loss": 0.0346, "grad_norm": 21.104068756103516, "learning_rate": 9.587878787878789e-06, "num_tokens": 282990.0, "completions/mean_length": 32.25, "completions/min_length": 22.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.6098703145980835, "rewards/meter/std": 0.3275175094604492, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6098703145980835, "rewards/total_composite/std": 0.3275175094604492, "reward": 0.6098703145980835, "reward_std": 0.3275175392627716, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15988147258758545, "sampling/sampling_logp_difference/max": 1.3653526306152344, "sampling/importance_sampling_ratio/min": 0.2552906572818756, "sampling/importance_sampling_ratio/mean": 0.9829218983650208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1308462470769882, "clip_ratio/low_mean": 0.037981835193932056, "clip_ratio/low_min": 0.037981835193932056, "clip_ratio/high_mean": 0.1296451175585389, "clip_ratio/high_max": 0.1296451175585389, "clip_ratio/region_mean": 0.16762695275247097, "reward_total_mean": 0.6098703145980835, "reward_meter_mean": 0.6098703145980835, "reward_meter_std": 0.3275175094604492, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6098703145980835, "reward_total_composite_std": 0.3275175094604492, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 137.0} {"timestamp_utc": "2026-04-11T19:37:51Z", "mode": "train", "global_step": 138, "epoch": 0.005329008341056534, "loss": 0.0749, "grad_norm": 19.138347625732422, "learning_rate": 9.584848484848486e-06, "num_tokens": 284471.0, "completions/mean_length": 27.125, "completions/min_length": 20.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.125, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.3993714153766632, "rewards/meter/std": 0.3401057720184326, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3993714153766632, "rewards/total_composite/std": 0.3401057720184326, "reward": 0.3993714153766632, "reward_std": 0.34010574221611023, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19733576476573944, "sampling/sampling_logp_difference/max": 0.8783740997314453, "sampling/importance_sampling_ratio/min": 0.4154578745365143, "sampling/importance_sampling_ratio/mean": 1.050313949584961, "sampling/importance_sampling_ratio/max": 1.970318078994751, "entropy": 2.293392300605774, "clip_ratio/low_mean": 0.11034664139151573, "clip_ratio/low_min": 0.11034664139151573, "clip_ratio/high_mean": 0.08452141657471657, "clip_ratio/high_max": 0.08452141657471657, "clip_ratio/region_mean": 0.1948680579662323, "reward_total_mean": 0.3993714153766632, "reward_meter_mean": 0.3993714153766632, "reward_meter_std": 0.3401057720184326, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3993714153766632, "reward_total_composite_std": 0.3401057720184326, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 138.0} {"timestamp_utc": "2026-04-11T19:37:59Z", "mode": "train", "global_step": 139, "epoch": 0.005367624343527958, "loss": -0.0421, "grad_norm": 4.607113838195801, "learning_rate": 9.581818181818181e-06, "num_tokens": 288591.0, "completions/mean_length": 292.0, "completions/min_length": 235.0, "completions/max_length": 335.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 292.0, "completions/min_terminated_length": 235.0, "completions/max_terminated_length": 335.0, "rewards/meter/mean": 0.8927704095840454, "rewards/meter/std": 0.13047616183757782, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.06681530922651291, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7447197437286377, "rewards/total_composite/std": 0.3232608437538147, "reward": 0.7447197437286377, "reward_std": 0.3232608437538147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19803562760353088, "sampling/sampling_logp_difference/max": 1.6082477569580078, "sampling/importance_sampling_ratio/min": 0.2002381682395935, "sampling/importance_sampling_ratio/mean": 1.0477176904678345, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.159882366657257, "clip_ratio/low_mean": 0.03914893604815006, "clip_ratio/low_min": 0.03914893604815006, "clip_ratio/high_mean": 0.13470745086669922, "clip_ratio/high_max": 0.13470745086669922, "clip_ratio/region_mean": 0.17385638691484928, "reward_total_mean": 0.7447197437286377, "reward_meter_mean": 0.8927704095840454, "reward_meter_std": 0.13047616183757782, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.06681530922651291, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7447197437286377, "reward_total_composite_std": 0.3232608437538147, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 139.0} {"timestamp_utc": "2026-04-11T19:38:08Z", "mode": "train", "global_step": 140, "epoch": 0.0054062403459993824, "loss": 0.0406, "grad_norm": 5.223753929138184, "learning_rate": 9.57878787878788e-06, "num_tokens": 292571.0, "completions/mean_length": 304.5, "completions/min_length": 205.0, "completions/max_length": 369.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 304.5, "completions/min_terminated_length": 205.0, "completions/max_terminated_length": 369.0, "rewards/meter/mean": 0.609697163105011, "rewards/meter/std": 0.17199508845806122, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.0534522607922554, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5782480835914612, "rewards/total_composite/std": 0.16025982797145844, "reward": 0.5782480835914612, "reward_std": 0.16025982797145844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19511224329471588, "sampling/sampling_logp_difference/max": 1.9568352699279785, "sampling/importance_sampling_ratio/min": 0.14130491018295288, "sampling/importance_sampling_ratio/mean": 1.034401535987854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.8953827619552612, "clip_ratio/low_mean": 0.084076764062047, "clip_ratio/low_min": 0.084076764062047, "clip_ratio/high_mean": 0.10052786208689213, "clip_ratio/high_max": 0.10052786208689213, "clip_ratio/region_mean": 0.18460462614893913, "reward_total_mean": 0.5782480835914612, "reward_meter_mean": 0.609697163105011, "reward_meter_std": 0.17199508845806122, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.0534522607922554, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5782480835914612, "reward_total_composite_std": 0.16025982797145844, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 140.0} {"timestamp_utc": "2026-04-11T19:38:13Z", "mode": "train", "global_step": 141, "epoch": 0.0054448563484708066, "loss": -0.0092, "grad_norm": 10.004317283630371, "learning_rate": 9.575757575757576e-06, "num_tokens": 294679.0, "completions/mean_length": 86.5, "completions/min_length": 65.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.5, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.414273202419281, "rewards/meter/std": 0.41314172744750977, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.414273202419281, "rewards/total_composite/std": 0.41314172744750977, "reward": 0.414273202419281, "reward_std": 0.41314172744750977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17779497802257538, "sampling/sampling_logp_difference/max": 1.0437870025634766, "sampling/importance_sampling_ratio/min": 0.35211867094039917, "sampling/importance_sampling_ratio/mean": 1.0314459800720215, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2329191118478775, "clip_ratio/low_mean": 0.08571217767894268, "clip_ratio/low_min": 0.08571217767894268, "clip_ratio/high_mean": 0.06570382788777351, "clip_ratio/high_max": 0.06570382788777351, "clip_ratio/region_mean": 0.1514160055667162, "reward_total_mean": 0.414273202419281, "reward_meter_mean": 0.414273202419281, "reward_meter_std": 0.41314172744750977, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.414273202419281, "reward_total_composite_std": 0.41314172744750977, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 141.0} {"timestamp_utc": "2026-04-11T19:38:18Z", "mode": "train", "global_step": 142, "epoch": 0.005483472350942231, "loss": -0.0499, "grad_norm": 13.359566688537598, "learning_rate": 9.572727272727273e-06, "num_tokens": 296372.0, "completions/mean_length": 47.625, "completions/min_length": 32.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.4189656376838684, "rewards/meter/std": 0.3781832754611969, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4189656376838684, "rewards/total_composite/std": 0.3781832754611969, "reward": 0.4189656376838684, "reward_std": 0.3781832456588745, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22827744483947754, "sampling/sampling_logp_difference/max": 1.9076499938964844, "sampling/importance_sampling_ratio/min": 0.1484287977218628, "sampling/importance_sampling_ratio/mean": 1.0515532493591309, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.629018932580948, "clip_ratio/low_mean": 0.11229986138641834, "clip_ratio/low_min": 0.11229986138641834, "clip_ratio/high_mean": 0.06938760355114937, "clip_ratio/high_max": 0.06938760355114937, "clip_ratio/region_mean": 0.1816874649375677, "reward_total_mean": 0.4189656376838684, "reward_meter_mean": 0.4189656376838684, "reward_meter_std": 0.3781832754611969, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4189656376838684, "reward_total_composite_std": 0.3781832754611969, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 142.0} {"timestamp_utc": "2026-04-11T19:38:23Z", "mode": "train", "global_step": 143, "epoch": 0.005522088353413655, "loss": -0.005, "grad_norm": 9.326652526855469, "learning_rate": 9.56969696969697e-06, "num_tokens": 298243.0, "completions/mean_length": 84.875, "completions/min_length": 67.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.5255526304244995, "rewards/meter/std": 0.3781281113624573, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5255526304244995, "rewards/total_composite/std": 0.3781281113624573, "reward": 0.5255526304244995, "reward_std": 0.3781280815601349, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20440031588077545, "sampling/sampling_logp_difference/max": 1.2985057830810547, "sampling/importance_sampling_ratio/min": 0.2729393243789673, "sampling/importance_sampling_ratio/mean": 1.0423548221588135, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4079796075820923, "clip_ratio/low_mean": 0.096000537276268, "clip_ratio/low_min": 0.096000537276268, "clip_ratio/high_mean": 0.0911912564188242, "clip_ratio/high_max": 0.0911912564188242, "clip_ratio/region_mean": 0.1871917936950922, "reward_total_mean": 0.5255526304244995, "reward_meter_mean": 0.5255526304244995, "reward_meter_std": 0.3781281113624573, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5255526304244995, "reward_total_composite_std": 0.3781281113624573, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 143.0} {"timestamp_utc": "2026-04-11T19:38:28Z", "mode": "train", "global_step": 144, "epoch": 0.005560704355885079, "loss": 0.1133, "grad_norm": 12.313883781433105, "learning_rate": 9.566666666666668e-06, "num_tokens": 299963.0, "completions/mean_length": 58.0, "completions/min_length": 41.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.6259543299674988, "rewards/meter/std": 0.40781956911087036, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6259543299674988, "rewards/total_composite/std": 0.40781956911087036, "reward": 0.6259543299674988, "reward_std": 0.40781959891319275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1947186291217804, "sampling/sampling_logp_difference/max": 1.2504253387451172, "sampling/importance_sampling_ratio/min": 0.2863829731941223, "sampling/importance_sampling_ratio/mean": 1.0449057817459106, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5128345042467117, "clip_ratio/low_mean": 0.09224239364266396, "clip_ratio/low_min": 0.09224239364266396, "clip_ratio/high_mean": 0.08965095691382885, "clip_ratio/high_max": 0.08965095691382885, "clip_ratio/region_mean": 0.1818933505564928, "reward_total_mean": 0.6259543299674988, "reward_meter_mean": 0.6259543299674988, "reward_meter_std": 0.40781956911087036, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6259543299674988, "reward_total_composite_std": 0.40781956911087036, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 144.0} {"timestamp_utc": "2026-04-11T19:38:33Z", "mode": "train", "global_step": 145, "epoch": 0.005599320358356503, "loss": -0.0099, "grad_norm": 9.220316886901855, "learning_rate": 9.563636363636365e-06, "num_tokens": 302259.0, "completions/mean_length": 108.0, "completions/min_length": 71.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.0, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.7156338095664978, "rewards/meter/std": 0.3621709942817688, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6794947385787964, "rewards/total_composite/std": 0.3623289465904236, "reward": 0.6794947385787964, "reward_std": 0.3623289465904236, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20363517105579376, "sampling/sampling_logp_difference/max": 1.6224696636199951, "sampling/importance_sampling_ratio/min": 0.19741055369377136, "sampling/importance_sampling_ratio/mean": 1.0389753580093384, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.594912603497505, "clip_ratio/low_mean": 0.07416711933910847, "clip_ratio/low_min": 0.07416711933910847, "clip_ratio/high_mean": 0.12736221216619015, "clip_ratio/high_max": 0.12736221216619015, "clip_ratio/region_mean": 0.20152933150529861, "reward_total_mean": 0.6794947385787964, "reward_meter_mean": 0.7156338095664978, "reward_meter_std": 0.3621709942817688, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6794947385787964, "reward_total_composite_std": 0.3623289465904236, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 145.0} {"timestamp_utc": "2026-04-11T19:38:38Z", "mode": "train", "global_step": 146, "epoch": 0.005637936360827927, "loss": 0.0802, "grad_norm": 21.128833770751953, "learning_rate": 9.56060606060606e-06, "num_tokens": 304531.0, "completions/mean_length": 85.0, "completions/min_length": 62.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.5389710068702698, "rewards/meter/std": 0.36102810502052307, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4409075975418091, "rewards/total_composite/std": 0.28550228476524353, "reward": 0.4409075975418091, "reward_std": 0.28550228476524353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20141077041625977, "sampling/sampling_logp_difference/max": 3.5244150161743164, "sampling/importance_sampling_ratio/min": 0.029469041153788567, "sampling/importance_sampling_ratio/mean": 1.0029895305633545, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9736258126795292, "clip_ratio/low_mean": 0.09793206304311752, "clip_ratio/low_min": 0.09793206304311752, "clip_ratio/high_mean": 0.06477605737745762, "clip_ratio/high_max": 0.06477605737745762, "clip_ratio/region_mean": 0.16270812042057514, "reward_total_mean": 0.4409075975418091, "reward_meter_mean": 0.5389710068702698, "reward_meter_std": 0.36102810502052307, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4409075975418091, "reward_total_composite_std": 0.28550228476524353, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 146.0} {"timestamp_utc": "2026-04-11T19:38:43Z", "mode": "train", "global_step": 147, "epoch": 0.005676552363299351, "loss": 0.0747, "grad_norm": 9.888257026672363, "learning_rate": 9.55757575757576e-06, "num_tokens": 306668.0, "completions/mean_length": 93.125, "completions/min_length": 68.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.7655929327011108, "rewards/meter/std": 0.3509427309036255, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6462192535400391, "rewards/total_composite/std": 0.43067827820777893, "reward": 0.6462192535400391, "reward_std": 0.43067827820777893, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19887900352478027, "sampling/sampling_logp_difference/max": 1.3454818725585938, "sampling/importance_sampling_ratio/min": 0.2604142129421234, "sampling/importance_sampling_ratio/mean": 1.0265202522277832, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3368784189224243, "clip_ratio/low_mean": 0.07888335548341274, "clip_ratio/low_min": 0.07888335548341274, "clip_ratio/high_mean": 0.11702838353812695, "clip_ratio/high_max": 0.11702838353812695, "clip_ratio/region_mean": 0.1959117390215397, "reward_total_mean": 0.6462192535400391, "reward_meter_mean": 0.7655929327011108, "reward_meter_std": 0.3509427309036255, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6462192535400391, "reward_total_composite_std": 0.43067827820777893, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 147.0} {"timestamp_utc": "2026-04-11T19:38:48Z", "mode": "train", "global_step": 148, "epoch": 0.005715168365770775, "loss": 0.0131, "grad_norm": 13.740280151367188, "learning_rate": 9.554545454545455e-06, "num_tokens": 308289.0, "completions/mean_length": 53.625, "completions/min_length": 35.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9531986117362976, "rewards/meter/std": 0.048368558287620544, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8362427353858948, "rewards/total_composite/std": 0.3412637710571289, "reward": 0.8362427353858948, "reward_std": 0.3412637710571289, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22549496591091156, "sampling/sampling_logp_difference/max": 2.545212745666504, "sampling/importance_sampling_ratio/min": 0.07845636457204819, "sampling/importance_sampling_ratio/mean": 1.0660349130630493, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.712312787771225, "clip_ratio/low_mean": 0.02163461595773697, "clip_ratio/low_min": 0.02163461595773697, "clip_ratio/high_mean": 0.18514032009989023, "clip_ratio/high_max": 0.18514032009989023, "clip_ratio/region_mean": 0.2067749360576272, "reward_total_mean": 0.8362427353858948, "reward_meter_mean": 0.9531986117362976, "reward_meter_std": 0.048368558287620544, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8362427353858948, "reward_total_composite_std": 0.3412637710571289, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 148.0} {"timestamp_utc": "2026-04-11T19:38:53Z", "mode": "train", "global_step": 149, "epoch": 0.005753784368242199, "loss": 0.1423, "grad_norm": 30.763076782226562, "learning_rate": 9.551515151515152e-06, "num_tokens": 310181.0, "completions/mean_length": 72.5, "completions/min_length": 65.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.5, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.764354407787323, "rewards/meter/std": 0.35320577025413513, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.764354407787323, "rewards/total_composite/std": 0.35320577025413513, "reward": 0.764354407787323, "reward_std": 0.35320577025413513, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14173762500286102, "sampling/sampling_logp_difference/max": 2.374579429626465, "sampling/importance_sampling_ratio/min": 0.09305361658334732, "sampling/importance_sampling_ratio/mean": 1.0017551183700562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.622876338660717, "clip_ratio/low_mean": 0.052138859406113625, "clip_ratio/low_min": 0.052138859406113625, "clip_ratio/high_mean": 0.06303809583187103, "clip_ratio/high_max": 0.06303809583187103, "clip_ratio/region_mean": 0.11517695523798466, "reward_total_mean": 0.764354407787323, "reward_meter_mean": 0.764354407787323, "reward_meter_std": 0.35320577025413513, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.764354407787323, "reward_total_composite_std": 0.35320577025413513, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 149.0} {"timestamp_utc": "2026-04-11T19:39:00Z", "mode": "train", "global_step": 150, "epoch": 0.0057924003707136235, "loss": -0.0588, "grad_norm": 5.072005748748779, "learning_rate": 9.54848484848485e-06, "num_tokens": 313791.0, "completions/mean_length": 255.25, "completions/min_length": 212.0, "completions/max_length": 319.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 255.25, "completions/min_terminated_length": 212.0, "completions/max_terminated_length": 319.0, "rewards/meter/mean": 0.4467889070510864, "rewards/meter/std": 0.29338762164115906, "rewards/count_adherence/mean": 0.9861111044883728, "rewards/count_adherence/std": 0.03928370773792267, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.32547223567962646, "rewards/total_composite/std": 0.2992675006389618, "reward": 0.32547223567962646, "reward_std": 0.2992675006389618, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20325259864330292, "sampling/sampling_logp_difference/max": 1.5813696384429932, "sampling/importance_sampling_ratio/min": 0.20569318532943726, "sampling/importance_sampling_ratio/mean": 1.0436980724334717, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1247769594192505, "clip_ratio/low_mean": 0.11349504999816418, "clip_ratio/low_min": 0.11349504999816418, "clip_ratio/high_mean": 0.06511373445391655, "clip_ratio/high_max": 0.06511373445391655, "clip_ratio/region_mean": 0.17860878445208073, "reward_total_mean": 0.32547223567962646, "reward_meter_mean": 0.4467889070510864, "reward_meter_std": 0.29338762164115906, "reward_count_adherence_mean": 0.9861111044883728, "reward_count_adherence_std": 0.03928370773792267, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.32547223567962646, "reward_total_composite_std": 0.2992675006389618, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 150.0} {"timestamp_utc": "2026-04-11T19:40:20Z", "mode": "eval", "global_step": 150, "epoch": 0.0057924003707136235, "eval_loss": NaN, "eval_runtime": 80.4379, "eval_samples_per_second": 1.293, "eval_steps_per_second": 0.162, "eval_num_tokens": 313791.0, "eval_completions/mean_length": 195.05769230769232, "eval_completions/min_length": 42.92307692307692, "eval_completions/max_length": 435.3076923076923, "eval_completions/clipped_ratio": 0.038461538461538464, "eval_completions/mean_terminated_length": 182.7815962571364, "eval_completions/min_terminated_length": 42.92307692307692, "eval_completions/max_terminated_length": 406.38461538461536, "eval_rewards/meter/mean": 0.45025052244846636, "eval_rewards/meter/std": 0.34984474113354314, "eval_rewards/count_adherence/mean": 0.9416975012192359, "eval_rewards/count_adherence/std": 0.08222452465158242, "eval_rewards/arabic_clean/mean": 0.9423076923076923, "eval_rewards/arabic_clean/std": 0.1631784851734455, "eval_rewards/total_composite/mean": 0.4012144311116292, "eval_rewards/total_composite/std": 0.3435493845206041, "eval_reward": 0.4012144311116292, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.14192955310528094, "eval_sampling/sampling_logp_difference/max": 1.3218139501718373, "eval_sampling/importance_sampling_ratio/min": 0.27021744961921984, "eval_sampling/importance_sampling_ratio/mean": 1.0403164625167847, "eval_sampling/importance_sampling_ratio/max": 1.593602758187514, "eval_entropy": 2.4764969715705285, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4012144311116292, "eval_reward_meter_mean": 0.45025052244846636, "eval_reward_meter_std": 0.34984474113354314, "eval_reward_count_adherence_mean": 0.9416975012192359, "eval_reward_count_adherence_std": 0.08222452465158242, "eval_reward_arabic_clean_mean": 0.9423076923076923, "eval_reward_arabic_clean_std": 0.1631784851734455, "eval_reward_total_composite_mean": 0.4012144311116292, "eval_reward_total_composite_std": 0.3435493845206041, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 150.0} {"timestamp_utc": "2026-04-11T19:40:31Z", "mode": "train", "global_step": 151, "epoch": 0.005831016373185048, "loss": -0.0419, "grad_norm": 4.5317206382751465, "learning_rate": 9.545454545454547e-06, "num_tokens": 318003.0, "completions/mean_length": 328.5, "completions/min_length": 272.0, "completions/max_length": 360.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 328.5, "completions/min_terminated_length": 272.0, "completions/max_terminated_length": 360.0, "rewards/meter/mean": 0.9837745428085327, "rewards/meter/std": 0.0232550036162138, "rewards/count_adherence/mean": 0.9722222089767456, "rewards/count_adherence/std": 0.05143444985151291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9565757513046265, "rewards/total_composite/std": 0.05796428769826889, "reward": 0.9565757513046265, "reward_std": 0.0579642690718174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18240459263324738, "sampling/sampling_logp_difference/max": 1.3508062362670898, "sampling/importance_sampling_ratio/min": 0.2590313255786896, "sampling/importance_sampling_ratio/mean": 1.0254464149475098, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.69022598862648, "clip_ratio/low_mean": 0.05626178905367851, "clip_ratio/low_min": 0.05626178905367851, "clip_ratio/high_mean": 0.10548157431185246, "clip_ratio/high_max": 0.10548157431185246, "clip_ratio/region_mean": 0.16174336336553097, "reward_total_mean": 0.9565757513046265, "reward_meter_mean": 0.9837745428085327, "reward_meter_std": 0.0232550036162138, "reward_count_adherence_mean": 0.9722222089767456, "reward_count_adherence_std": 0.05143444985151291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9565757513046265, "reward_total_composite_std": 0.05796428769826889, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 151.0} {"timestamp_utc": "2026-04-11T19:40:39Z", "mode": "train", "global_step": 152, "epoch": 0.005869632375656472, "loss": 0.0225, "grad_norm": 6.531708240509033, "learning_rate": 9.542424242424242e-06, "num_tokens": 321511.0, "completions/mean_length": 222.5, "completions/min_length": 196.0, "completions/max_length": 260.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 222.5, "completions/min_terminated_length": 196.0, "completions/max_terminated_length": 260.0, "rewards/meter/mean": 0.6115673780441284, "rewards/meter/std": 0.26747360825538635, "rewards/count_adherence/mean": 0.984375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6029950380325317, "rewards/total_composite/std": 0.2708563804626465, "reward": 0.6029950380325317, "reward_std": 0.2708563804626465, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20755378901958466, "sampling/sampling_logp_difference/max": 1.5887994766235352, "sampling/importance_sampling_ratio/min": 0.20417055487632751, "sampling/importance_sampling_ratio/mean": 1.0410678386688232, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6781843453645706, "clip_ratio/low_mean": 0.08865037560462952, "clip_ratio/low_min": 0.08865037560462952, "clip_ratio/high_mean": 0.10017775557935238, "clip_ratio/high_max": 0.10017775557935238, "clip_ratio/region_mean": 0.1888281311839819, "reward_total_mean": 0.6029950380325317, "reward_meter_mean": 0.6115673780441284, "reward_meter_std": 0.26747360825538635, "reward_count_adherence_mean": 0.984375, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6029950380325317, "reward_total_composite_std": 0.2708563804626465, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 152.0} {"timestamp_utc": "2026-04-11T19:40:43Z", "mode": "train", "global_step": 153, "epoch": 0.005908248378127896, "loss": 0.1928, "grad_norm": 15.966812133789062, "learning_rate": 9.539393939393941e-06, "num_tokens": 323205.0, "completions/mean_length": 50.75, "completions/min_length": 37.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7692294716835022, "rewards/meter/std": 0.3993333876132965, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7692294716835022, "rewards/total_composite/std": 0.3993333876132965, "reward": 0.7692294716835022, "reward_std": 0.3993334174156189, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21327464282512665, "sampling/sampling_logp_difference/max": 1.4604730606079102, "sampling/importance_sampling_ratio/min": 0.23212644457817078, "sampling/importance_sampling_ratio/mean": 1.0662130117416382, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6396835446357727, "clip_ratio/low_mean": 0.05802265927195549, "clip_ratio/low_min": 0.05802265927195549, "clip_ratio/high_mean": 0.15194394160062075, "clip_ratio/high_max": 0.15194394160062075, "clip_ratio/region_mean": 0.20996660087257624, "reward_total_mean": 0.7692294716835022, "reward_meter_mean": 0.7692294716835022, "reward_meter_std": 0.3993333876132965, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7692294716835022, "reward_total_composite_std": 0.3993333876132965, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 153.0} {"timestamp_utc": "2026-04-11T19:40:48Z", "mode": "train", "global_step": 154, "epoch": 0.005946864380599321, "loss": -0.069, "grad_norm": 9.946159362792969, "learning_rate": 9.536363636363637e-06, "num_tokens": 324984.0, "completions/mean_length": 72.375, "completions/min_length": 57.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.375, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.7901514172554016, "rewards/meter/std": 0.3216681182384491, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.666408896446228, "rewards/total_composite/std": 0.4116549491882324, "reward": 0.666408896446228, "reward_std": 0.4116549491882324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20986607670783997, "sampling/sampling_logp_difference/max": 1.975529670715332, "sampling/importance_sampling_ratio/min": 0.1386878341436386, "sampling/importance_sampling_ratio/mean": 1.0114688873291016, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7783928215503693, "clip_ratio/low_mean": 0.047645075246691704, "clip_ratio/low_min": 0.047645075246691704, "clip_ratio/high_mean": 0.14308988489210606, "clip_ratio/high_max": 0.14308988489210606, "clip_ratio/region_mean": 0.19073496013879776, "reward_total_mean": 0.666408896446228, "reward_meter_mean": 0.7901514172554016, "reward_meter_std": 0.3216681182384491, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.666408896446228, "reward_total_composite_std": 0.4116549491882324, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 154.0} {"timestamp_utc": "2026-04-11T19:40:55Z", "mode": "train", "global_step": 155, "epoch": 0.005985480383070745, "loss": -0.0834, "grad_norm": 5.41533899307251, "learning_rate": 9.533333333333334e-06, "num_tokens": 328524.0, "completions/mean_length": 235.5, "completions/min_length": 126.0, "completions/max_length": 276.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 235.5, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 276.0, "rewards/meter/mean": 0.7103838920593262, "rewards/meter/std": 0.29926469922065735, "rewards/count_adherence/mean": 0.9821428656578064, "rewards/count_adherence/std": 0.05050762742757797, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6477268934249878, "rewards/total_composite/std": 0.3904498517513275, "reward": 0.6477268934249878, "reward_std": 0.3904498219490051, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19282367825508118, "sampling/sampling_logp_difference/max": 2.1544742584228516, "sampling/importance_sampling_ratio/min": 0.1159641370177269, "sampling/importance_sampling_ratio/mean": 1.0317730903625488, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7660705894231796, "clip_ratio/low_mean": 0.06773262470960617, "clip_ratio/low_min": 0.06773262470960617, "clip_ratio/high_mean": 0.08849271759390831, "clip_ratio/high_max": 0.08849271759390831, "clip_ratio/region_mean": 0.15622534230351448, "reward_total_mean": 0.6477268934249878, "reward_meter_mean": 0.7103838920593262, "reward_meter_std": 0.29926469922065735, "reward_count_adherence_mean": 0.9821428656578064, "reward_count_adherence_std": 0.05050762742757797, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6477268934249878, "reward_total_composite_std": 0.3904498517513275, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 155.0} {"timestamp_utc": "2026-04-11T19:41:01Z", "mode": "train", "global_step": 156, "epoch": 0.006024096385542169, "loss": 0.0523, "grad_norm": 8.701234817504883, "learning_rate": 9.530303030303031e-06, "num_tokens": 330785.0, "completions/mean_length": 105.625, "completions/min_length": 90.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.625, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.6397774815559387, "rewards/meter/std": 0.31731224060058594, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5999155044555664, "rewards/total_composite/std": 0.2907305359840393, "reward": 0.5999155044555664, "reward_std": 0.2907305359840393, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16831091046333313, "sampling/sampling_logp_difference/max": 1.6831111907958984, "sampling/importance_sampling_ratio/min": 0.18579503893852234, "sampling/importance_sampling_ratio/mean": 1.0126723051071167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4543365985155106, "clip_ratio/low_mean": 0.05361151322722435, "clip_ratio/low_min": 0.05361151322722435, "clip_ratio/high_mean": 0.08997475728392601, "clip_ratio/high_max": 0.08997475728392601, "clip_ratio/region_mean": 0.14358627051115036, "reward_total_mean": 0.5999155044555664, "reward_meter_mean": 0.6397774815559387, "reward_meter_std": 0.31731224060058594, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5999155044555664, "reward_total_composite_std": 0.2907305359840393, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 156.0} {"timestamp_utc": "2026-04-11T19:41:06Z", "mode": "train", "global_step": 157, "epoch": 0.006062712388013593, "loss": 0.0172, "grad_norm": 9.539356231689453, "learning_rate": 9.527272727272729e-06, "num_tokens": 332526.0, "completions/mean_length": 62.625, "completions/min_length": 45.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.625, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8717015385627747, "rewards/meter/std": 0.322393000125885, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8717015385627747, "rewards/total_composite/std": 0.322393000125885, "reward": 0.8717015385627747, "reward_std": 0.322393000125885, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18296034634113312, "sampling/sampling_logp_difference/max": 1.5920934677124023, "sampling/importance_sampling_ratio/min": 0.20349915325641632, "sampling/importance_sampling_ratio/mean": 1.0461691617965698, "sampling/importance_sampling_ratio/max": 1.7814452648162842, "entropy": 2.5809740722179413, "clip_ratio/low_mean": 0.026209676638245583, "clip_ratio/low_min": 0.026209676638245583, "clip_ratio/high_mean": 0.14061870239675045, "clip_ratio/high_max": 0.14061870239675045, "clip_ratio/region_mean": 0.16682837903499603, "reward_total_mean": 0.8717015385627747, "reward_meter_mean": 0.8717015385627747, "reward_meter_std": 0.322393000125885, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8717015385627747, "reward_total_composite_std": 0.322393000125885, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 157.0} {"timestamp_utc": "2026-04-11T19:41:11Z", "mode": "train", "global_step": 158, "epoch": 0.006101328390485017, "loss": 0.1023, "grad_norm": 13.795117378234863, "learning_rate": 9.524242424242424e-06, "num_tokens": 334145.0, "completions/mean_length": 32.375, "completions/min_length": 21.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.375, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.86896812915802, "rewards/meter/std": 0.2910270094871521, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.86896812915802, "rewards/total_composite/std": 0.2910270094871521, "reward": 0.86896812915802, "reward_std": 0.2910270094871521, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19816143810749054, "sampling/sampling_logp_difference/max": 1.4168579578399658, "sampling/importance_sampling_ratio/min": 0.24247469007968903, "sampling/importance_sampling_ratio/mean": 1.0380147695541382, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.62221197783947, "clip_ratio/low_mean": 0.03427810175344348, "clip_ratio/low_min": 0.03427810175344348, "clip_ratio/high_mean": 0.11192020168527961, "clip_ratio/high_max": 0.11192020168527961, "clip_ratio/region_mean": 0.1461983034387231, "reward_total_mean": 0.86896812915802, "reward_meter_mean": 0.86896812915802, "reward_meter_std": 0.2910270094871521, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.86896812915802, "reward_total_composite_std": 0.2910270094871521, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 158.0} {"timestamp_utc": "2026-04-11T19:41:15Z", "mode": "train", "global_step": 159, "epoch": 0.006139944392956441, "loss": 0.0781, "grad_norm": 11.608582496643066, "learning_rate": 9.521212121212121e-06, "num_tokens": 335861.0, "completions/mean_length": 56.5, "completions/min_length": 47.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6755526065826416, "rewards/meter/std": 0.40379220247268677, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6755526065826416, "rewards/total_composite/std": 0.40379220247268677, "reward": 0.6755526065826416, "reward_std": 0.40379223227500916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1799980252981186, "sampling/sampling_logp_difference/max": 1.1121330261230469, "sampling/importance_sampling_ratio/min": 0.32885676622390747, "sampling/importance_sampling_ratio/mean": 1.0527642965316772, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.072466492652893, "clip_ratio/low_mean": 0.043226663023233414, "clip_ratio/low_min": 0.043226663023233414, "clip_ratio/high_mean": 0.13957421202212572, "clip_ratio/high_max": 0.13957421202212572, "clip_ratio/region_mean": 0.18280087504535913, "reward_total_mean": 0.6755526065826416, "reward_meter_mean": 0.6755526065826416, "reward_meter_std": 0.40379220247268677, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6755526065826416, "reward_total_composite_std": 0.40379220247268677, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 159.0} {"timestamp_utc": "2026-04-11T19:41:20Z", "mode": "train", "global_step": 160, "epoch": 0.006178560395427865, "loss": -0.1254, "grad_norm": 9.607455253601074, "learning_rate": 9.518181818181819e-06, "num_tokens": 337794.0, "completions/mean_length": 81.625, "completions/min_length": 57.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.625, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.5209656357765198, "rewards/meter/std": 0.3587253987789154, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5209656357765198, "rewards/total_composite/std": 0.3587253987789154, "reward": 0.5209656357765198, "reward_std": 0.358725368976593, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20390784740447998, "sampling/sampling_logp_difference/max": 1.0271854400634766, "sampling/importance_sampling_ratio/min": 0.35801321268081665, "sampling/importance_sampling_ratio/mean": 1.049707293510437, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.947203129529953, "clip_ratio/low_mean": 0.07780237961560488, "clip_ratio/low_min": 0.07780237961560488, "clip_ratio/high_mean": 0.08933841064572334, "clip_ratio/high_max": 0.08933841064572334, "clip_ratio/region_mean": 0.16714079026132822, "reward_total_mean": 0.5209656357765198, "reward_meter_mean": 0.5209656357765198, "reward_meter_std": 0.3587253987789154, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5209656357765198, "reward_total_composite_std": 0.3587253987789154, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 160.0} {"timestamp_utc": "2026-04-11T19:41:27Z", "mode": "train", "global_step": 161, "epoch": 0.0062171763978992895, "loss": -0.0557, "grad_norm": 6.381593704223633, "learning_rate": 9.515151515151516e-06, "num_tokens": 340718.0, "completions/mean_length": 173.5, "completions/min_length": 139.0, "completions/max_length": 206.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 173.5, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 206.0, "rewards/meter/mean": 0.8734483122825623, "rewards/meter/std": 0.19228114187717438, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8628791570663452, "rewards/total_composite/std": 0.2208014875650406, "reward": 0.8628791570663452, "reward_std": 0.2208014875650406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20080114901065826, "sampling/sampling_logp_difference/max": 1.1833868026733398, "sampling/importance_sampling_ratio/min": 0.3062398135662079, "sampling/importance_sampling_ratio/mean": 1.044055700302124, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.2587929368019104, "clip_ratio/low_mean": 0.0405045822262764, "clip_ratio/low_min": 0.0405045822262764, "clip_ratio/high_mean": 0.15176175348460674, "clip_ratio/high_max": 0.15176175348460674, "clip_ratio/region_mean": 0.19226633571088314, "reward_total_mean": 0.8628791570663452, "reward_meter_mean": 0.8734483122825623, "reward_meter_std": 0.19228114187717438, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8628791570663452, "reward_total_composite_std": 0.2208014875650406, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 161.0} {"timestamp_utc": "2026-04-11T19:41:31Z", "mode": "train", "global_step": 162, "epoch": 0.006255792400370714, "loss": -0.0145, "grad_norm": 12.352027893066406, "learning_rate": 9.512121212121213e-06, "num_tokens": 342611.0, "completions/mean_length": 76.625, "completions/min_length": 64.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.625, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.7090976238250732, "rewards/meter/std": 0.351256400346756, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7090976238250732, "rewards/total_composite/std": 0.351256400346756, "reward": 0.7090976238250732, "reward_std": 0.3512563705444336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2120855152606964, "sampling/sampling_logp_difference/max": 1.341689109802246, "sampling/importance_sampling_ratio/min": 0.2703118920326233, "sampling/importance_sampling_ratio/mean": 1.0393997430801392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3186896294355392, "clip_ratio/low_mean": 0.07329358533024788, "clip_ratio/low_min": 0.07329358533024788, "clip_ratio/high_mean": 0.15742158330976963, "clip_ratio/high_max": 0.15742158330976963, "clip_ratio/region_mean": 0.2307151686400175, "reward_total_mean": 0.7090976238250732, "reward_meter_mean": 0.7090976238250732, "reward_meter_std": 0.351256400346756, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7090976238250732, "reward_total_composite_std": 0.351256400346756, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 162.0} {"timestamp_utc": "2026-04-11T19:41:41Z", "mode": "train", "global_step": 163, "epoch": 0.006294408402842138, "loss": 0.0554, "grad_norm": 3.614668846130371, "learning_rate": 9.50909090909091e-06, "num_tokens": 347883.0, "completions/mean_length": 451.0, "completions/min_length": 392.0, "completions/max_length": 504.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 451.0, "completions/min_terminated_length": 392.0, "completions/max_terminated_length": 504.0, "rewards/meter/mean": 0.8174892663955688, "rewards/meter/std": 0.28463214635849, "rewards/count_adherence/mean": 0.8181818127632141, "rewards/count_adherence/std": 0.06872081756591797, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.5725052356719971, "rewards/total_composite/std": 0.36880064010620117, "reward": 0.5725052356719971, "reward_std": 0.3688006103038788, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1946493536233902, "sampling/sampling_logp_difference/max": 2.3504714965820312, "sampling/importance_sampling_ratio/min": 0.09532421082258224, "sampling/importance_sampling_ratio/mean": 1.045919418334961, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1186185479164124, "clip_ratio/low_mean": 0.04785826615989208, "clip_ratio/low_min": 0.04785826615989208, "clip_ratio/high_mean": 0.09156403131783009, "clip_ratio/high_max": 0.09156403131783009, "clip_ratio/region_mean": 0.13942229747772217, "reward_total_mean": 0.5725052356719971, "reward_meter_mean": 0.8174892663955688, "reward_meter_std": 0.28463214635849, "reward_count_adherence_mean": 0.8181818127632141, "reward_count_adherence_std": 0.06872081756591797, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.5725052356719971, "reward_total_composite_std": 0.36880064010620117, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 163.0} {"timestamp_utc": "2026-04-11T19:41:46Z", "mode": "train", "global_step": 164, "epoch": 0.006333024405313562, "loss": 0.0587, "grad_norm": 10.339271545410156, "learning_rate": 9.506060606060606e-06, "num_tokens": 349747.0, "completions/mean_length": 65.0, "completions/min_length": 48.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.47777289152145386, "rewards/meter/std": 0.44221097230911255, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.47777289152145386, "rewards/total_composite/std": 0.44221097230911255, "reward": 0.47777289152145386, "reward_std": 0.44221097230911255, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2009221911430359, "sampling/sampling_logp_difference/max": 1.2063875198364258, "sampling/importance_sampling_ratio/min": 0.2992764711380005, "sampling/importance_sampling_ratio/mean": 1.0429495573043823, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.003392845392227, "clip_ratio/low_mean": 0.11853201128542423, "clip_ratio/low_min": 0.11853201128542423, "clip_ratio/high_mean": 0.09482496604323387, "clip_ratio/high_max": 0.09482496604323387, "clip_ratio/region_mean": 0.2133569773286581, "reward_total_mean": 0.47777289152145386, "reward_meter_mean": 0.47777289152145386, "reward_meter_std": 0.44221097230911255, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.47777289152145386, "reward_total_composite_std": 0.44221097230911255, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 164.0} {"timestamp_utc": "2026-04-11T19:41:50Z", "mode": "train", "global_step": 165, "epoch": 0.006371640407784986, "loss": 0.0353, "grad_norm": 13.097195625305176, "learning_rate": 9.503030303030303e-06, "num_tokens": 351519.0, "completions/mean_length": 61.5, "completions/min_length": 54.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.5606287121772766, "rewards/meter/std": 0.3949960768222809, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5606287121772766, "rewards/total_composite/std": 0.3949960768222809, "reward": 0.5606287121772766, "reward_std": 0.39499610662460327, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18365828692913055, "sampling/sampling_logp_difference/max": 1.585710048675537, "sampling/importance_sampling_ratio/min": 0.20480231940746307, "sampling/importance_sampling_ratio/mean": 1.0260679721832275, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.010311871767044, "clip_ratio/low_mean": 0.08252524584531784, "clip_ratio/low_min": 0.08252524584531784, "clip_ratio/high_mean": 0.07877860497683287, "clip_ratio/high_max": 0.07877860497683287, "clip_ratio/region_mean": 0.1613038508221507, "reward_total_mean": 0.5606287121772766, "reward_meter_mean": 0.5606287121772766, "reward_meter_std": 0.3949960768222809, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5606287121772766, "reward_total_composite_std": 0.3949960768222809, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 165.0} {"timestamp_utc": "2026-04-11T19:41:56Z", "mode": "train", "global_step": 166, "epoch": 0.00641025641025641, "loss": -0.0261, "grad_norm": 7.968050956726074, "learning_rate": 9.5e-06, "num_tokens": 353796.0, "completions/mean_length": 123.625, "completions/min_length": 99.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.625, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.37855178117752075, "rewards/meter/std": 0.23558726906776428, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.35550376772880554, "rewards/total_composite/std": 0.26453739404678345, "reward": 0.35550376772880554, "reward_std": 0.26453739404678345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23242639005184174, "sampling/sampling_logp_difference/max": 1.4124670028686523, "sampling/importance_sampling_ratio/min": 0.24354171752929688, "sampling/importance_sampling_ratio/mean": 1.0575261116027832, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.55351784825325, "clip_ratio/low_mean": 0.0659086387604475, "clip_ratio/low_min": 0.0659086387604475, "clip_ratio/high_mean": 0.13404671289026737, "clip_ratio/high_max": 0.13404671289026737, "clip_ratio/region_mean": 0.19995535165071487, "reward_total_mean": 0.35550376772880554, "reward_meter_mean": 0.37855178117752075, "reward_meter_std": 0.23558726906776428, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.35550376772880554, "reward_total_composite_std": 0.26453739404678345, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 166.0} {"timestamp_utc": "2026-04-11T19:42:01Z", "mode": "train", "global_step": 167, "epoch": 0.006448872412727834, "loss": 0.128, "grad_norm": 12.225397109985352, "learning_rate": 9.496969696969698e-06, "num_tokens": 355489.0, "completions/mean_length": 53.625, "completions/min_length": 35.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.541317880153656, "rewards/meter/std": 0.35997992753982544, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.541317880153656, "rewards/total_composite/std": 0.35997992753982544, "reward": 0.541317880153656, "reward_std": 0.35997989773750305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19463400542736053, "sampling/sampling_logp_difference/max": 1.56561279296875, "sampling/importance_sampling_ratio/min": 0.2710648775100708, "sampling/importance_sampling_ratio/mean": 1.0362927913665771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5232774317264557, "clip_ratio/low_mean": 0.06835224945098162, "clip_ratio/low_min": 0.06835224945098162, "clip_ratio/high_mean": 0.11530855856835842, "clip_ratio/high_max": 0.11530855856835842, "clip_ratio/region_mean": 0.18366080801934004, "reward_total_mean": 0.541317880153656, "reward_meter_mean": 0.541317880153656, "reward_meter_std": 0.35997992753982544, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.541317880153656, "reward_total_composite_std": 0.35997992753982544, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 167.0} {"timestamp_utc": "2026-04-11T19:42:06Z", "mode": "train", "global_step": 168, "epoch": 0.006487488415199258, "loss": 0.0174, "grad_norm": 9.189430236816406, "learning_rate": 9.493939393939395e-06, "num_tokens": 357276.0, "completions/mean_length": 72.375, "completions/min_length": 57.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.375, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.521911084651947, "rewards/meter/std": 0.40068677067756653, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.521911084651947, "rewards/total_composite/std": 0.40068677067756653, "reward": 0.521911084651947, "reward_std": 0.40068677067756653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19903360307216644, "sampling/sampling_logp_difference/max": 1.3080291748046875, "sampling/importance_sampling_ratio/min": 0.2703523635864258, "sampling/importance_sampling_ratio/mean": 1.0414764881134033, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.740328684449196, "clip_ratio/low_mean": 0.08560106623917818, "clip_ratio/low_min": 0.08560106623917818, "clip_ratio/high_mean": 0.08096018619835377, "clip_ratio/high_max": 0.08096018619835377, "clip_ratio/region_mean": 0.16656125243753195, "reward_total_mean": 0.521911084651947, "reward_meter_mean": 0.521911084651947, "reward_meter_std": 0.40068677067756653, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.521911084651947, "reward_total_composite_std": 0.40068677067756653, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 168.0} {"timestamp_utc": "2026-04-11T19:42:10Z", "mode": "train", "global_step": 169, "epoch": 0.006526104417670682, "loss": 0.2465, "grad_norm": 13.522438049316406, "learning_rate": 9.490909090909092e-06, "num_tokens": 358907.0, "completions/mean_length": 47.875, "completions/min_length": 27.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.875, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.4249013066291809, "rewards/meter/std": 0.4389844834804535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4249013066291809, "rewards/total_composite/std": 0.4389844834804535, "reward": 0.4249013066291809, "reward_std": 0.4389844834804535, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2200341522693634, "sampling/sampling_logp_difference/max": 1.1910114288330078, "sampling/importance_sampling_ratio/min": 0.3039137125015259, "sampling/importance_sampling_ratio/mean": 1.0431842803955078, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0451967269182205, "clip_ratio/low_mean": 0.08651185780763626, "clip_ratio/low_min": 0.08651185780763626, "clip_ratio/high_mean": 0.12107311747968197, "clip_ratio/high_max": 0.12107311747968197, "clip_ratio/region_mean": 0.20758497528731823, "reward_total_mean": 0.4249013066291809, "reward_meter_mean": 0.4249013066291809, "reward_meter_std": 0.4389844834804535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4249013066291809, "reward_total_composite_std": 0.4389844834804535, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 169.0} {"timestamp_utc": "2026-04-11T19:42:14Z", "mode": "train", "global_step": 170, "epoch": 0.006564720420142107, "loss": 0.0794, "grad_norm": 16.293785095214844, "learning_rate": 9.487878787878788e-06, "num_tokens": 360430.0, "completions/mean_length": 33.375, "completions/min_length": 23.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.375, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.46223390102386475, "rewards/meter/std": 0.4552871286869049, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.33747631311416626, "rewards/total_composite/std": 0.423090398311615, "reward": 0.33747631311416626, "reward_std": 0.423090398311615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21340948343276978, "sampling/sampling_logp_difference/max": 1.056483268737793, "sampling/importance_sampling_ratio/min": 0.3476763367652893, "sampling/importance_sampling_ratio/mean": 1.0465396642684937, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.248675227165222, "clip_ratio/low_mean": 0.0997670404613018, "clip_ratio/low_min": 0.0997670404613018, "clip_ratio/high_mean": 0.08509290590882301, "clip_ratio/high_max": 0.08509290590882301, "clip_ratio/region_mean": 0.18485994637012482, "reward_total_mean": 0.33747631311416626, "reward_meter_mean": 0.46223390102386475, "reward_meter_std": 0.4552871286869049, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.33747631311416626, "reward_total_composite_std": 0.423090398311615, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 170.0} {"timestamp_utc": "2026-04-11T19:42:23Z", "mode": "train", "global_step": 171, "epoch": 0.006603336422613531, "loss": 0.0339, "grad_norm": 3.9159412384033203, "learning_rate": 9.484848484848485e-06, "num_tokens": 364930.0, "completions/mean_length": 372.5, "completions/min_length": 346.0, "completions/max_length": 393.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 372.5, "completions/min_terminated_length": 346.0, "completions/max_terminated_length": 393.0, "rewards/meter/mean": 0.8391759395599365, "rewards/meter/std": 0.14736339449882507, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.05143444612622261, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6796014308929443, "rewards/total_composite/std": 0.3177638053894043, "reward": 0.6796014308929443, "reward_std": 0.3177638053894043, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18074527382850647, "sampling/sampling_logp_difference/max": 1.5130558013916016, "sampling/importance_sampling_ratio/min": 0.2202359437942505, "sampling/importance_sampling_ratio/mean": 1.0409563779830933, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7181632220745087, "clip_ratio/low_mean": 0.045382389798760414, "clip_ratio/low_min": 0.045382389798760414, "clip_ratio/high_mean": 0.09890040010213852, "clip_ratio/high_max": 0.09890040010213852, "clip_ratio/region_mean": 0.14428278990089893, "reward_total_mean": 0.6796014308929443, "reward_meter_mean": 0.8391759395599365, "reward_meter_std": 0.14736339449882507, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.05143444612622261, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6796014308929443, "reward_total_composite_std": 0.3177638053894043, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 171.0} {"timestamp_utc": "2026-04-11T19:42:28Z", "mode": "train", "global_step": 172, "epoch": 0.0066419524250849555, "loss": -0.0949, "grad_norm": 10.551653861999512, "learning_rate": 9.481818181818182e-06, "num_tokens": 367070.0, "completions/mean_length": 93.5, "completions/min_length": 61.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.3134363889694214, "rewards/meter/std": 0.3427402675151825, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3134363889694214, "rewards/total_composite/std": 0.3427402675151825, "reward": 0.3134363889694214, "reward_std": 0.3427402675151825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.209889754652977, "sampling/sampling_logp_difference/max": 1.237997055053711, "sampling/importance_sampling_ratio/min": 0.2899644374847412, "sampling/importance_sampling_ratio/mean": 1.0271856784820557, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3088208436965942, "clip_ratio/low_mean": 0.13063165172934532, "clip_ratio/low_min": 0.13063165172934532, "clip_ratio/high_mean": 0.04081328213214874, "clip_ratio/high_max": 0.04081328213214874, "clip_ratio/region_mean": 0.17144493386149406, "reward_total_mean": 0.3134363889694214, "reward_meter_mean": 0.3134363889694214, "reward_meter_std": 0.3427402675151825, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3134363889694214, "reward_total_composite_std": 0.3427402675151825, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 172.0} {"timestamp_utc": "2026-04-11T19:42:38Z", "mode": "train", "global_step": 173, "epoch": 0.00668056842755638, "loss": 0.1304, "grad_norm": 3.27034068107605, "learning_rate": 9.47878787878788e-06, "num_tokens": 371854.0, "completions/mean_length": 451.0, "completions/min_length": 316.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 442.2857360839844, "completions/min_terminated_length": 316.0, "completions/max_terminated_length": 504.0, "rewards/meter/mean": 0.6058165431022644, "rewards/meter/std": 0.27364325523376465, "rewards/count_adherence/mean": 0.9204545617103577, "rewards/count_adherence/std": 0.058260902762413025, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.43490633368492126, "rewards/total_composite/std": 0.33899492025375366, "reward": 0.43490633368492126, "reward_std": 0.3389948904514313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19972197711467743, "sampling/sampling_logp_difference/max": 1.4985361099243164, "sampling/importance_sampling_ratio/min": 0.22345705330371857, "sampling/importance_sampling_ratio/mean": 1.0502684116363525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9574913382530212, "clip_ratio/low_mean": 0.05308797210454941, "clip_ratio/low_min": 0.05308797210454941, "clip_ratio/high_mean": 0.06190796010196209, "clip_ratio/high_max": 0.06190796010196209, "clip_ratio/region_mean": 0.1149959322065115, "reward_total_mean": 0.43490633368492126, "reward_meter_mean": 0.6058165431022644, "reward_meter_std": 0.27364325523376465, "reward_count_adherence_mean": 0.9204545617103577, "reward_count_adherence_std": 0.058260902762413025, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.43490633368492126, "reward_total_composite_std": 0.33899492025375366, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 173.0} {"timestamp_utc": "2026-04-11T19:42:43Z", "mode": "train", "global_step": 174, "epoch": 0.006719184430027804, "loss": 0.0781, "grad_norm": 11.655610084533691, "learning_rate": 9.475757575757577e-06, "num_tokens": 373685.0, "completions/mean_length": 55.875, "completions/min_length": 41.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7013630867004395, "rewards/meter/std": 0.37237003445625305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7013630867004395, "rewards/total_composite/std": 0.37237003445625305, "reward": 0.7013630867004395, "reward_std": 0.37237003445625305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19981878995895386, "sampling/sampling_logp_difference/max": 1.8281879425048828, "sampling/importance_sampling_ratio/min": 0.16070450842380524, "sampling/importance_sampling_ratio/mean": 1.0189765691757202, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.167596936225891, "clip_ratio/low_mean": 0.05789176933467388, "clip_ratio/low_min": 0.05789176933467388, "clip_ratio/high_mean": 0.13539652153849602, "clip_ratio/high_max": 0.13539652153849602, "clip_ratio/region_mean": 0.1932882908731699, "reward_total_mean": 0.7013630867004395, "reward_meter_mean": 0.7013630867004395, "reward_meter_std": 0.37237003445625305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7013630867004395, "reward_total_composite_std": 0.37237003445625305, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 174.0} {"timestamp_utc": "2026-04-11T19:42:48Z", "mode": "train", "global_step": 175, "epoch": 0.006757800432499228, "loss": -0.0845, "grad_norm": 8.652692794799805, "learning_rate": 9.472727272727274e-06, "num_tokens": 375783.0, "completions/mean_length": 99.25, "completions/min_length": 69.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.25, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.8306900858879089, "rewards/meter/std": 0.2705877721309662, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8306900858879089, "rewards/total_composite/std": 0.2705877721309662, "reward": 0.8306900858879089, "reward_std": 0.2705877721309662, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20954445004463196, "sampling/sampling_logp_difference/max": 1.618638515472412, "sampling/importance_sampling_ratio/min": 0.1981683075428009, "sampling/importance_sampling_ratio/mean": 1.0498780012130737, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0380934178829193, "clip_ratio/low_mean": 0.05846922844648361, "clip_ratio/low_min": 0.05846922844648361, "clip_ratio/high_mean": 0.1139191659167409, "clip_ratio/high_max": 0.1139191659167409, "clip_ratio/region_mean": 0.1723883943632245, "reward_total_mean": 0.8306900858879089, "reward_meter_mean": 0.8306900858879089, "reward_meter_std": 0.2705877721309662, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8306900858879089, "reward_total_composite_std": 0.2705877721309662, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 175.0} {"timestamp_utc": "2026-04-11T19:42:53Z", "mode": "train", "global_step": 176, "epoch": 0.006796416434970652, "loss": 0.0503, "grad_norm": 18.11098861694336, "learning_rate": 9.469696969696971e-06, "num_tokens": 377613.0, "completions/mean_length": 55.75, "completions/min_length": 46.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.75, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.5684157609939575, "rewards/meter/std": 0.3562447726726532, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5684157609939575, "rewards/total_composite/std": 0.3562447726726532, "reward": 0.5684157609939575, "reward_std": 0.3562447726726532, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18087293207645416, "sampling/sampling_logp_difference/max": 1.8970457315444946, "sampling/importance_sampling_ratio/min": 0.15001113712787628, "sampling/importance_sampling_ratio/mean": 0.9817885756492615, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1572269052267075, "clip_ratio/low_mean": 0.07028611842542887, "clip_ratio/low_min": 0.07028611842542887, "clip_ratio/high_mean": 0.10677518881857395, "clip_ratio/high_max": 0.10677518881857395, "clip_ratio/region_mean": 0.17706130724400282, "reward_total_mean": 0.5684157609939575, "reward_meter_mean": 0.5684157609939575, "reward_meter_std": 0.3562447726726532, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5684157609939575, "reward_total_composite_std": 0.3562447726726532, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 176.0} {"timestamp_utc": "2026-04-11T19:42:58Z", "mode": "train", "global_step": 177, "epoch": 0.006835032437442076, "loss": 0.0261, "grad_norm": 11.097681999206543, "learning_rate": 9.466666666666667e-06, "num_tokens": 379410.0, "completions/mean_length": 65.625, "completions/min_length": 56.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.625, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.6728242039680481, "rewards/meter/std": 0.35246020555496216, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6728242039680481, "rewards/total_composite/std": 0.35246020555496216, "reward": 0.6728242039680481, "reward_std": 0.35246017575263977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1819072812795639, "sampling/sampling_logp_difference/max": 1.5235271453857422, "sampling/importance_sampling_ratio/min": 0.21794182062149048, "sampling/importance_sampling_ratio/mean": 1.0345712900161743, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.025869235396385, "clip_ratio/low_mean": 0.07266573421657085, "clip_ratio/low_min": 0.07266573421657085, "clip_ratio/high_mean": 0.10677864961326122, "clip_ratio/high_max": 0.10677864961326122, "clip_ratio/region_mean": 0.17944438382983208, "reward_total_mean": 0.6728242039680481, "reward_meter_mean": 0.6728242039680481, "reward_meter_std": 0.35246020555496216, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6728242039680481, "reward_total_composite_std": 0.35246020555496216, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 177.0} {"timestamp_utc": "2026-04-11T19:43:03Z", "mode": "train", "global_step": 178, "epoch": 0.0068736484399135, "loss": 0.1003, "grad_norm": 8.50960636138916, "learning_rate": 9.463636363636364e-06, "num_tokens": 381389.0, "completions/mean_length": 95.375, "completions/min_length": 54.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.375, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.2792768180370331, "rewards/meter/std": 0.3471120297908783, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.2792768180370331, "rewards/total_composite/std": 0.3471120297908783, "reward": 0.2792768180370331, "reward_std": 0.3471120595932007, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18758107721805573, "sampling/sampling_logp_difference/max": 1.1053590774536133, "sampling/importance_sampling_ratio/min": 0.331091970205307, "sampling/importance_sampling_ratio/mean": 1.0582326650619507, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3764619678258896, "clip_ratio/low_mean": 0.1264551393687725, "clip_ratio/low_min": 0.1264551393687725, "clip_ratio/high_mean": 0.053513072431087494, "clip_ratio/high_max": 0.053513072431087494, "clip_ratio/region_mean": 0.17996821179986, "reward_total_mean": 0.2792768180370331, "reward_meter_mean": 0.2792768180370331, "reward_meter_std": 0.3471120297908783, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.2792768180370331, "reward_total_composite_std": 0.3471120297908783, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 178.0} {"timestamp_utc": "2026-04-11T19:43:07Z", "mode": "train", "global_step": 179, "epoch": 0.006912264442384924, "loss": 0.0501, "grad_norm": 13.35677433013916, "learning_rate": 9.460606060606061e-06, "num_tokens": 382901.0, "completions/mean_length": 39.0, "completions/min_length": 22.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.742904782295227, "rewards/meter/std": 0.4480074644088745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.742904782295227, "rewards/total_composite/std": 0.4480074644088745, "reward": 0.742904782295227, "reward_std": 0.4480074346065521, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2040482759475708, "sampling/sampling_logp_difference/max": 1.5499753952026367, "sampling/importance_sampling_ratio/min": 0.21225318312644958, "sampling/importance_sampling_ratio/mean": 1.0228500366210938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.807357758283615, "clip_ratio/low_mean": 0.05450693517923355, "clip_ratio/low_min": 0.05450693517923355, "clip_ratio/high_mean": 0.15819299593567848, "clip_ratio/high_max": 0.15819299593567848, "clip_ratio/region_mean": 0.21269993111491203, "reward_total_mean": 0.742904782295227, "reward_meter_mean": 0.742904782295227, "reward_meter_std": 0.4480074644088745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.742904782295227, "reward_total_composite_std": 0.4480074644088745, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 179.0} {"timestamp_utc": "2026-04-11T19:43:12Z", "mode": "train", "global_step": 180, "epoch": 0.006950880444856348, "loss": -0.0462, "grad_norm": 7.980728626251221, "learning_rate": 9.457575757575759e-06, "num_tokens": 385212.0, "completions/mean_length": 118.875, "completions/min_length": 91.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.875, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.6352285146713257, "rewards/meter/std": 0.3728267252445221, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.4976061284542084, "rewards/total_composite/std": 0.40696853399276733, "reward": 0.4976061284542084, "reward_std": 0.40696853399276733, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20611317455768585, "sampling/sampling_logp_difference/max": 1.2493963241577148, "sampling/importance_sampling_ratio/min": 0.2866778075695038, "sampling/importance_sampling_ratio/mean": 1.0363506078720093, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9363907277584076, "clip_ratio/low_mean": 0.10372502729296684, "clip_ratio/low_min": 0.10372502729296684, "clip_ratio/high_mean": 0.0775204561650753, "clip_ratio/high_max": 0.0775204561650753, "clip_ratio/region_mean": 0.18124548345804214, "reward_total_mean": 0.4976061284542084, "reward_meter_mean": 0.6352285146713257, "reward_meter_std": 0.3728267252445221, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.4976061284542084, "reward_total_composite_std": 0.40696853399276733, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 180.0} {"timestamp_utc": "2026-04-11T19:43:18Z", "mode": "train", "global_step": 181, "epoch": 0.0069894964473277725, "loss": 0.0206, "grad_norm": 5.923155784606934, "learning_rate": 9.454545454545456e-06, "num_tokens": 388081.0, "completions/mean_length": 174.625, "completions/min_length": 141.0, "completions/max_length": 204.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 174.625, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 204.0, "rewards/meter/mean": 0.6006770133972168, "rewards/meter/std": 0.31185153126716614, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5388948917388916, "rewards/total_composite/std": 0.3470546007156372, "reward": 0.5388948917388916, "reward_std": 0.3470546305179596, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2052806168794632, "sampling/sampling_logp_difference/max": 1.3783597946166992, "sampling/importance_sampling_ratio/min": 0.25199154019355774, "sampling/importance_sampling_ratio/mean": 1.0360268354415894, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1949117481708527, "clip_ratio/low_mean": 0.051841133274137974, "clip_ratio/low_min": 0.051841133274137974, "clip_ratio/high_mean": 0.11146864108741283, "clip_ratio/high_max": 0.11146864108741283, "clip_ratio/region_mean": 0.1633097743615508, "reward_total_mean": 0.5388948917388916, "reward_meter_mean": 0.6006770133972168, "reward_meter_std": 0.31185153126716614, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5388948917388916, "reward_total_composite_std": 0.3470546007156372, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 181.0} {"timestamp_utc": "2026-04-11T19:43:23Z", "mode": "train", "global_step": 182, "epoch": 0.007028112449799197, "loss": -0.0381, "grad_norm": 8.314396858215332, "learning_rate": 9.451515151515153e-06, "num_tokens": 390055.0, "completions/mean_length": 84.75, "completions/min_length": 52.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.75, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.8241270780563354, "rewards/meter/std": 0.21255002915859222, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8241270780563354, "rewards/total_composite/std": 0.21255002915859222, "reward": 0.8241270780563354, "reward_std": 0.21255002915859222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1786930412054062, "sampling/sampling_logp_difference/max": 1.5513114929199219, "sampling/importance_sampling_ratio/min": 0.211969792842865, "sampling/importance_sampling_ratio/mean": 1.0361807346343994, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.674738720059395, "clip_ratio/low_mean": 0.06412622984498739, "clip_ratio/low_min": 0.06412622984498739, "clip_ratio/high_mean": 0.1037982078269124, "clip_ratio/high_max": 0.1037982078269124, "clip_ratio/region_mean": 0.1679244376718998, "reward_total_mean": 0.8241270780563354, "reward_meter_mean": 0.8241270780563354, "reward_meter_std": 0.21255002915859222, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8241270780563354, "reward_total_composite_std": 0.21255002915859222, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 182.0} {"timestamp_utc": "2026-04-11T19:43:34Z", "mode": "train", "global_step": 183, "epoch": 0.007066728452270621, "loss": -0.2128, "grad_norm": 3.2879745960235596, "learning_rate": 9.448484848484849e-06, "num_tokens": 393147.0, "completions/mean_length": 485.5, "completions/min_length": 396.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 441.3333435058594, "completions/min_terminated_length": 396.0, "completions/max_terminated_length": 486.0, "rewards/meter/mean": 0.26494020223617554, "rewards/meter/std": 0.2154451161623001, "rewards/count_adherence/mean": 0.7767857313156128, "rewards/count_adherence/std": 0.21407301723957062, "rewards/arabic_clean/mean": 0.125, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.023662419989705086, "rewards/total_composite/std": 0.06692743301391602, "reward": 0.023662419989705086, "reward_std": 0.06692743301391602, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22992289066314697, "sampling/sampling_logp_difference/max": 1.410196304321289, "sampling/importance_sampling_ratio/min": 0.24409537017345428, "sampling/importance_sampling_ratio/mean": 1.0776352882385254, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6914453506469727, "clip_ratio/low_mean": 0.031273381784558296, "clip_ratio/low_min": 0.031273381784558296, "clip_ratio/high_mean": 0.016119910404086113, "clip_ratio/high_max": 0.016119910404086113, "clip_ratio/region_mean": 0.04739329218864441, "reward_total_mean": 0.023662419989705086, "reward_meter_mean": 0.26494020223617554, "reward_meter_std": 0.2154451161623001, "reward_count_adherence_mean": 0.7767857313156128, "reward_count_adherence_std": 0.21407301723957062, "reward_arabic_clean_mean": 0.125, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.023662419989705086, "reward_total_composite_std": 0.06692743301391602, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 183.0} {"timestamp_utc": "2026-04-11T19:43:41Z", "mode": "train", "global_step": 184, "epoch": 0.007105344454742045, "loss": 0.0335, "grad_norm": 7.487835884094238, "learning_rate": 9.445454545454546e-06, "num_tokens": 395894.0, "completions/mean_length": 151.375, "completions/min_length": 109.0, "completions/max_length": 199.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.375, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 199.0, "rewards/meter/mean": 0.7551294565200806, "rewards/meter/std": 0.2453630119562149, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7357177734375, "rewards/total_composite/std": 0.23533061146736145, "reward": 0.7357177734375, "reward_std": 0.23533061146736145, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22830148041248322, "sampling/sampling_logp_difference/max": 1.888480544090271, "sampling/importance_sampling_ratio/min": 0.1513015180826187, "sampling/importance_sampling_ratio/mean": 1.0623719692230225, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.480381518602371, "clip_ratio/low_mean": 0.06921044550836086, "clip_ratio/low_min": 0.06921044550836086, "clip_ratio/high_mean": 0.12044411525130272, "clip_ratio/high_max": 0.12044411525130272, "clip_ratio/region_mean": 0.18965456075966358, "reward_total_mean": 0.7357177734375, "reward_meter_mean": 0.7551294565200806, "reward_meter_std": 0.2453630119562149, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7357177734375, "reward_total_composite_std": 0.23533061146736145, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 184.0} {"timestamp_utc": "2026-04-11T19:43:51Z", "mode": "train", "global_step": 185, "epoch": 0.007143960457213469, "loss": -0.2352, "grad_norm": 1.9176102876663208, "learning_rate": 9.442424242424243e-06, "num_tokens": 398431.0, "completions/mean_length": 485.125, "completions/min_length": 375.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.75, "completions/mean_terminated_length": 404.5, "completions/min_terminated_length": 375.0, "completions/max_terminated_length": 434.0, "rewards/meter/mean": 0.6249135732650757, "rewards/meter/std": 0.15311351418495178, "rewards/count_adherence/mean": 0.7647058963775635, "rewards/count_adherence/std": 0.2037706971168518, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.36697742342948914, "rewards/total_composite/std": 0.280710369348526, "reward": 0.36697742342948914, "reward_std": 0.280710369348526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22530639171600342, "sampling/sampling_logp_difference/max": 1.2742891311645508, "sampling/importance_sampling_ratio/min": 0.2796296775341034, "sampling/importance_sampling_ratio/mean": 1.0640817880630493, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.180066168308258, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.04156912490725517, "clip_ratio/high_max": 0.04156912490725517, "clip_ratio/region_mean": 0.04156912490725517, "reward_total_mean": 0.36697742342948914, "reward_meter_mean": 0.6249135732650757, "reward_meter_std": 0.15311351418495178, "reward_count_adherence_mean": 0.7647058963775635, "reward_count_adherence_std": 0.2037706971168518, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.36697742342948914, "reward_total_composite_std": 0.280710369348526, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 185.0} {"timestamp_utc": "2026-04-11T19:43:57Z", "mode": "train", "global_step": 186, "epoch": 0.007182576459684893, "loss": -0.0388, "grad_norm": 13.209904670715332, "learning_rate": 9.43939393939394e-06, "num_tokens": 400845.0, "completions/mean_length": 118.75, "completions/min_length": 74.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.75, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.4079614281654358, "rewards/meter/std": 0.2506245970726013, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625821024179459, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.33415091037750244, "rewards/total_composite/std": 0.2877470552921295, "reward": 0.33415091037750244, "reward_std": 0.2877470552921295, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25658828020095825, "sampling/sampling_logp_difference/max": 2.9441909790039062, "sampling/importance_sampling_ratio/min": 0.05264463275671005, "sampling/importance_sampling_ratio/mean": 1.018273115158081, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.863920621573925, "clip_ratio/low_mean": 0.14540163800120354, "clip_ratio/low_min": 0.14540163800120354, "clip_ratio/high_mean": 0.08783877268433571, "clip_ratio/high_max": 0.08783877268433571, "clip_ratio/region_mean": 0.23324041068553925, "reward_total_mean": 0.33415091037750244, "reward_meter_mean": 0.4079614281654358, "reward_meter_std": 0.2506245970726013, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625821024179459, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.33415091037750244, "reward_total_composite_std": 0.2877470552921295, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 186.0} {"timestamp_utc": "2026-04-11T19:44:03Z", "mode": "train", "global_step": 187, "epoch": 0.007221192462156318, "loss": 0.0416, "grad_norm": 6.4664082527160645, "learning_rate": 9.436363636363636e-06, "num_tokens": 403557.0, "completions/mean_length": 163.0, "completions/min_length": 155.0, "completions/max_length": 175.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 163.0, "completions/min_terminated_length": 155.0, "completions/max_terminated_length": 175.0, "rewards/meter/mean": 0.582764744758606, "rewards/meter/std": 0.3457765281200409, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5632787346839905, "rewards/total_composite/std": 0.3654843270778656, "reward": 0.5632787346839905, "reward_std": 0.3654842972755432, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1593349426984787, "sampling/sampling_logp_difference/max": 1.3233938217163086, "sampling/importance_sampling_ratio/min": 0.2662302553653717, "sampling/importance_sampling_ratio/mean": 1.0383338928222656, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.991146519780159, "clip_ratio/low_mean": 0.08290823642164469, "clip_ratio/low_min": 0.08290823642164469, "clip_ratio/high_mean": 0.05676225572824478, "clip_ratio/high_max": 0.05676225572824478, "clip_ratio/region_mean": 0.13967049214988947, "reward_total_mean": 0.5632787346839905, "reward_meter_mean": 0.582764744758606, "reward_meter_std": 0.3457765281200409, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5632787346839905, "reward_total_composite_std": 0.3654843270778656, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 187.0} {"timestamp_utc": "2026-04-11T19:44:08Z", "mode": "train", "global_step": 188, "epoch": 0.007259808464627742, "loss": 0.0815, "grad_norm": 13.624017715454102, "learning_rate": 9.433333333333335e-06, "num_tokens": 405714.0, "completions/mean_length": 93.625, "completions/min_length": 51.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.6261414885520935, "rewards/meter/std": 0.19565246999263763, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5144075155258179, "rewards/total_composite/std": 0.2748170793056488, "reward": 0.5144075155258179, "reward_std": 0.2748170793056488, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19976428151130676, "sampling/sampling_logp_difference/max": 1.8644428253173828, "sampling/importance_sampling_ratio/min": 0.1549825519323349, "sampling/importance_sampling_ratio/mean": 1.026123285293579, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5476962476968765, "clip_ratio/low_mean": 0.04282732866704464, "clip_ratio/low_min": 0.04282732866704464, "clip_ratio/high_mean": 0.1269478127360344, "clip_ratio/high_max": 0.1269478127360344, "clip_ratio/region_mean": 0.16977514140307903, "reward_total_mean": 0.5144075155258179, "reward_meter_mean": 0.6261414885520935, "reward_meter_std": 0.19565246999263763, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5144075155258179, "reward_total_composite_std": 0.2748170793056488, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 188.0} {"timestamp_utc": "2026-04-11T19:44:12Z", "mode": "train", "global_step": 189, "epoch": 0.007298424467099166, "loss": 0.0495, "grad_norm": 11.650201797485352, "learning_rate": 9.43030303030303e-06, "num_tokens": 407329.0, "completions/mean_length": 55.875, "completions/min_length": 47.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.5849356055259705, "rewards/meter/std": 0.39748692512512207, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5849356055259705, "rewards/total_composite/std": 0.39748692512512207, "reward": 0.5849356055259705, "reward_std": 0.39748692512512207, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16425088047981262, "sampling/sampling_logp_difference/max": 1.3752784729003906, "sampling/importance_sampling_ratio/min": 0.25276920199394226, "sampling/importance_sampling_ratio/mean": 1.0523135662078857, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7971455752849579, "clip_ratio/low_mean": 0.08321618381887674, "clip_ratio/low_min": 0.08321618381887674, "clip_ratio/high_mean": 0.07970546465367079, "clip_ratio/high_max": 0.07970546465367079, "clip_ratio/region_mean": 0.16292164847254753, "reward_total_mean": 0.5849356055259705, "reward_meter_mean": 0.5849356055259705, "reward_meter_std": 0.39748692512512207, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5849356055259705, "reward_total_composite_std": 0.39748692512512207, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 189.0} {"timestamp_utc": "2026-04-11T19:44:17Z", "mode": "train", "global_step": 190, "epoch": 0.00733704046957059, "loss": 0.1149, "grad_norm": 10.552634239196777, "learning_rate": 9.427272727272728e-06, "num_tokens": 409327.0, "completions/mean_length": 69.75, "completions/min_length": 58.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.6343156695365906, "rewards/meter/std": 0.4757094979286194, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6193662881851196, "rewards/total_composite/std": 0.4956565201282501, "reward": 0.6193662881851196, "reward_std": 0.4956565201282501, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20032422244548798, "sampling/sampling_logp_difference/max": 1.4165172576904297, "sampling/importance_sampling_ratio/min": 0.24255730211734772, "sampling/importance_sampling_ratio/mean": 1.0550477504730225, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6996882259845734, "clip_ratio/low_mean": 0.049503629095852375, "clip_ratio/low_min": 0.049503629095852375, "clip_ratio/high_mean": 0.08876706939190626, "clip_ratio/high_max": 0.08876706939190626, "clip_ratio/region_mean": 0.13827069848775864, "reward_total_mean": 0.6193662881851196, "reward_meter_mean": 0.6343156695365906, "reward_meter_std": 0.4757094979286194, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6193662881851196, "reward_total_composite_std": 0.4956565201282501, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 190.0} {"timestamp_utc": "2026-04-11T19:44:25Z", "mode": "train", "global_step": 191, "epoch": 0.007375656472042014, "loss": 0.0501, "grad_norm": 4.086877822875977, "learning_rate": 9.424242424242425e-06, "num_tokens": 413518.0, "completions/mean_length": 323.875, "completions/min_length": 305.0, "completions/max_length": 356.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 323.875, "completions/min_terminated_length": 305.0, "completions/max_terminated_length": 356.0, "rewards/meter/mean": 0.8315606117248535, "rewards/meter/std": 0.19411730766296387, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8315606117248535, "rewards/total_composite/std": 0.19411730766296387, "reward": 0.8315606117248535, "reward_std": 0.19411730766296387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1796930879354477, "sampling/sampling_logp_difference/max": 1.630167007446289, "sampling/importance_sampling_ratio/min": 0.19589684903621674, "sampling/importance_sampling_ratio/mean": 1.0462325811386108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.755393832921982, "clip_ratio/low_mean": 0.029843377880752087, "clip_ratio/low_min": 0.029843377880752087, "clip_ratio/high_mean": 0.10810728743672371, "clip_ratio/high_max": 0.10810728743672371, "clip_ratio/region_mean": 0.1379506653174758, "reward_total_mean": 0.8315606117248535, "reward_meter_mean": 0.8315606117248535, "reward_meter_std": 0.19411730766296387, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8315606117248535, "reward_total_composite_std": 0.19411730766296387, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 191.0} {"timestamp_utc": "2026-04-11T19:44:29Z", "mode": "train", "global_step": 192, "epoch": 0.0074142724745134385, "loss": 0.153, "grad_norm": 18.393901824951172, "learning_rate": 9.421212121212122e-06, "num_tokens": 414921.0, "completions/mean_length": 27.375, "completions/min_length": 19.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.375, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9494917392730713, "rewards/meter/std": 0.08201512694358826, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9494917392730713, "rewards/total_composite/std": 0.08201512694358826, "reward": 0.9494917392730713, "reward_std": 0.08201511204242706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1969519406557083, "sampling/sampling_logp_difference/max": 1.4734296798706055, "sampling/importance_sampling_ratio/min": 0.22913827002048492, "sampling/importance_sampling_ratio/mean": 1.0469014644622803, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.018664211034775, "clip_ratio/low_mean": 0.01315789483487606, "clip_ratio/low_min": 0.01315789483487606, "clip_ratio/high_mean": 0.16411421447992325, "clip_ratio/high_max": 0.16411421447992325, "clip_ratio/region_mean": 0.1772721093147993, "reward_total_mean": 0.9494917392730713, "reward_meter_mean": 0.9494917392730713, "reward_meter_std": 0.08201512694358826, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9494917392730713, "reward_total_composite_std": 0.08201512694358826, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 192.0} {"timestamp_utc": "2026-04-11T19:44:34Z", "mode": "train", "global_step": 193, "epoch": 0.007452888476984863, "loss": 0.0298, "grad_norm": 8.817331314086914, "learning_rate": 9.418181818181818e-06, "num_tokens": 416883.0, "completions/mean_length": 95.25, "completions/min_length": 69.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.25, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.6673023104667664, "rewards/meter/std": 0.2610948383808136, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6673023104667664, "rewards/total_composite/std": 0.2610948383808136, "reward": 0.6673023104667664, "reward_std": 0.2610948085784912, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18318572640419006, "sampling/sampling_logp_difference/max": 1.5481929779052734, "sampling/importance_sampling_ratio/min": 0.21263186633586884, "sampling/importance_sampling_ratio/mean": 1.0500800609588623, "sampling/importance_sampling_ratio/max": 1.9678956270217896, "entropy": 2.2031291872262955, "clip_ratio/low_mean": 0.07560309767723083, "clip_ratio/low_min": 0.07560309767723083, "clip_ratio/high_mean": 0.07883104495704174, "clip_ratio/high_max": 0.07883104495704174, "clip_ratio/region_mean": 0.15443414263427258, "reward_total_mean": 0.6673023104667664, "reward_meter_mean": 0.6673023104667664, "reward_meter_std": 0.2610948383808136, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6673023104667664, "reward_total_composite_std": 0.2610948383808136, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 193.0} {"timestamp_utc": "2026-04-11T19:44:39Z", "mode": "train", "global_step": 194, "epoch": 0.007491504479456287, "loss": -0.0527, "grad_norm": 9.746423721313477, "learning_rate": 9.415151515151515e-06, "num_tokens": 418806.0, "completions/mean_length": 66.375, "completions/min_length": 41.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.7032723426818848, "rewards/meter/std": 0.39979878067970276, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5814144611358643, "rewards/total_composite/std": 0.45054084062576294, "reward": 0.5814144611358643, "reward_std": 0.45054084062576294, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1895114779472351, "sampling/sampling_logp_difference/max": 1.1117887496948242, "sampling/importance_sampling_ratio/min": 0.3289699852466583, "sampling/importance_sampling_ratio/mean": 1.0394792556762695, "sampling/importance_sampling_ratio/max": 1.9423279762268066, "entropy": 2.68495112657547, "clip_ratio/low_mean": 0.09499397501349449, "clip_ratio/low_min": 0.09499397501349449, "clip_ratio/high_mean": 0.10805463790893555, "clip_ratio/high_max": 0.10805463790893555, "clip_ratio/region_mean": 0.20304861292243004, "reward_total_mean": 0.5814144611358643, "reward_meter_mean": 0.7032723426818848, "reward_meter_std": 0.39979878067970276, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5814144611358643, "reward_total_composite_std": 0.45054084062576294, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 194.0} {"timestamp_utc": "2026-04-11T19:44:43Z", "mode": "train", "global_step": 195, "epoch": 0.007530120481927711, "loss": 0.1632, "grad_norm": 22.751644134521484, "learning_rate": 9.412121212121212e-06, "num_tokens": 420399.0, "completions/mean_length": 33.125, "completions/min_length": 26.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.5200808048248291, "rewards/meter/std": 0.43741583824157715, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5200808048248291, "rewards/total_composite/std": 0.43741583824157715, "reward": 0.5200808048248291, "reward_std": 0.43741583824157715, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2031438797712326, "sampling/sampling_logp_difference/max": 3.59218692779541, "sampling/importance_sampling_ratio/min": 0.027538040652871132, "sampling/importance_sampling_ratio/mean": 1.015283226966858, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.359821267426014, "clip_ratio/low_mean": 0.09933342039585114, "clip_ratio/low_min": 0.09933342039585114, "clip_ratio/high_mean": 0.054079256020486355, "clip_ratio/high_max": 0.054079256020486355, "clip_ratio/region_mean": 0.1534126764163375, "reward_total_mean": 0.5200808048248291, "reward_meter_mean": 0.5200808048248291, "reward_meter_std": 0.43741583824157715, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5200808048248291, "reward_total_composite_std": 0.43741583824157715, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 195.0} {"timestamp_utc": "2026-04-11T19:44:48Z", "mode": "train", "global_step": 196, "epoch": 0.007568736484399135, "loss": 0.0894, "grad_norm": 14.902490615844727, "learning_rate": 9.40909090909091e-06, "num_tokens": 422210.0, "completions/mean_length": 61.375, "completions/min_length": 40.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.6579214334487915, "rewards/meter/std": 0.34652820229530334, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6541217565536499, "rewards/total_composite/std": 0.3544676899909973, "reward": 0.6541217565536499, "reward_std": 0.3544676601886749, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19266894459724426, "sampling/sampling_logp_difference/max": 1.1338281631469727, "sampling/importance_sampling_ratio/min": 0.32179901003837585, "sampling/importance_sampling_ratio/mean": 1.0256552696228027, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.188107267022133, "clip_ratio/low_mean": 0.04345930367708206, "clip_ratio/low_min": 0.04345930367708206, "clip_ratio/high_mean": 0.13002329412847757, "clip_ratio/high_max": 0.13002329412847757, "clip_ratio/region_mean": 0.17348259780555964, "reward_total_mean": 0.6541217565536499, "reward_meter_mean": 0.6579214334487915, "reward_meter_std": 0.34652820229530334, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6541217565536499, "reward_total_composite_std": 0.3544676899909973, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 196.0} {"timestamp_utc": "2026-04-11T19:44:53Z", "mode": "train", "global_step": 197, "epoch": 0.007607352486870559, "loss": -0.0611, "grad_norm": 8.72506332397461, "learning_rate": 9.406060606060607e-06, "num_tokens": 423961.0, "completions/mean_length": 73.875, "completions/min_length": 61.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.875, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.7866647243499756, "rewards/meter/std": 0.33512723445892334, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7566624879837036, "rewards/total_composite/std": 0.3962303102016449, "reward": 0.7566624879837036, "reward_std": 0.3962303400039673, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16485705971717834, "sampling/sampling_logp_difference/max": 1.4514455795288086, "sampling/importance_sampling_ratio/min": 0.23423144221305847, "sampling/importance_sampling_ratio/mean": 1.0307003259658813, "sampling/importance_sampling_ratio/max": 1.8759212493896484, "entropy": 2.119014173746109, "clip_ratio/low_mean": 0.03474697098135948, "clip_ratio/low_min": 0.03474697098135948, "clip_ratio/high_mean": 0.1163166556507349, "clip_ratio/high_max": 0.1163166556507349, "clip_ratio/region_mean": 0.15106362663209438, "reward_total_mean": 0.7566624879837036, "reward_meter_mean": 0.7866647243499756, "reward_meter_std": 0.33512723445892334, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7566624879837036, "reward_total_composite_std": 0.3962303102016449, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 197.0} {"timestamp_utc": "2026-04-11T19:44:58Z", "mode": "train", "global_step": 198, "epoch": 0.007645968489341983, "loss": 0.0372, "grad_norm": 8.083459854125977, "learning_rate": 9.403030303030304e-06, "num_tokens": 425728.0, "completions/mean_length": 72.875, "completions/min_length": 56.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.8608720302581787, "rewards/meter/std": 0.3337464928627014, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8608720302581787, "rewards/total_composite/std": 0.3337464928627014, "reward": 0.8608720302581787, "reward_std": 0.33374646306037903, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.181260883808136, "sampling/sampling_logp_difference/max": 1.2895793914794922, "sampling/importance_sampling_ratio/min": 0.27538660168647766, "sampling/importance_sampling_ratio/mean": 1.0441133975982666, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.166542023420334, "clip_ratio/low_mean": 0.02056962065398693, "clip_ratio/low_min": 0.02056962065398693, "clip_ratio/high_mean": 0.11953294090926647, "clip_ratio/high_max": 0.11953294090926647, "clip_ratio/region_mean": 0.1401025615632534, "reward_total_mean": 0.8608720302581787, "reward_meter_mean": 0.8608720302581787, "reward_meter_std": 0.3337464928627014, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8608720302581787, "reward_total_composite_std": 0.3337464928627014, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 198.0} {"timestamp_utc": "2026-04-11T19:45:03Z", "mode": "train", "global_step": 199, "epoch": 0.007684584491813407, "loss": 0.0447, "grad_norm": 6.834539890289307, "learning_rate": 9.4e-06, "num_tokens": 428283.0, "completions/mean_length": 140.375, "completions/min_length": 133.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.375, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.8502488136291504, "rewards/meter/std": 0.2549903690814972, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8193603754043579, "rewards/total_composite/std": 0.250792533159256, "reward": 0.8193603754043579, "reward_std": 0.250792533159256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17601756751537323, "sampling/sampling_logp_difference/max": 1.9291143417358398, "sampling/importance_sampling_ratio/min": 0.14527681469917297, "sampling/importance_sampling_ratio/mean": 1.0295875072479248, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1282119899988174, "clip_ratio/low_mean": 0.0449892720207572, "clip_ratio/low_min": 0.0449892720207572, "clip_ratio/high_mean": 0.11455865763127804, "clip_ratio/high_max": 0.11455865763127804, "clip_ratio/region_mean": 0.15954792965203524, "reward_total_mean": 0.8193603754043579, "reward_meter_mean": 0.8502488136291504, "reward_meter_std": 0.2549903690814972, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8193603754043579, "reward_total_composite_std": 0.250792533159256, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 199.0} {"timestamp_utc": "2026-04-11T19:45:09Z", "mode": "train", "global_step": 200, "epoch": 0.007723200494284831, "loss": 0.0566, "grad_norm": 14.159527778625488, "learning_rate": 9.396969696969697e-06, "num_tokens": 430070.0, "completions/mean_length": 55.375, "completions/min_length": 36.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.375, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8980928659439087, "rewards/meter/std": 0.20199771225452423, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8980928659439087, "rewards/total_composite/std": 0.20199771225452423, "reward": 0.8980928659439087, "reward_std": 0.20199769735336304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17297667264938354, "sampling/sampling_logp_difference/max": 1.4214141368865967, "sampling/importance_sampling_ratio/min": 0.24137245118618011, "sampling/importance_sampling_ratio/mean": 1.031834602355957, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2072382867336273, "clip_ratio/low_mean": 0.010593220591545105, "clip_ratio/low_min": 0.010593220591545105, "clip_ratio/high_mean": 0.14806674886494875, "clip_ratio/high_max": 0.14806674886494875, "clip_ratio/region_mean": 0.15865996945649385, "reward_total_mean": 0.8980928659439087, "reward_meter_mean": 0.8980928659439087, "reward_meter_std": 0.20199771225452423, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8980928659439087, "reward_total_composite_std": 0.20199771225452423, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 200.0} {"timestamp_utc": "2026-04-11T19:46:34Z", "mode": "eval", "global_step": 200, "epoch": 0.007723200494284831, "eval_loss": NaN, "eval_runtime": 84.7428, "eval_samples_per_second": 1.227, "eval_steps_per_second": 0.153, "eval_num_tokens": 430070.0, "eval_completions/mean_length": 217.46153846153845, "eval_completions/min_length": 56.0, "eval_completions/max_length": 457.3076923076923, "eval_completions/clipped_ratio": 0.07692307692307693, "eval_completions/mean_terminated_length": 192.05220383864182, "eval_completions/min_terminated_length": 56.0, "eval_completions/max_terminated_length": 368.0, "eval_rewards/meter/mean": 0.4592640560406905, "eval_rewards/meter/std": 0.3293467943484967, "eval_rewards/count_adherence/mean": 0.9569021555093619, "eval_rewards/count_adherence/std": 0.07193636994522351, "eval_rewards/arabic_clean/mean": 0.9423076923076923, "eval_rewards/arabic_clean/std": 0.12560976010102493, "eval_rewards/total_composite/mean": 0.4151367247104645, "eval_rewards/total_composite/std": 0.3096239452178662, "eval_reward": 0.4151367247104645, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.1303755704026956, "eval_sampling/sampling_logp_difference/max": 1.260967914874737, "eval_sampling/importance_sampling_ratio/min": 0.28642855469997114, "eval_sampling/importance_sampling_ratio/mean": 1.0360724650896513, "eval_sampling/importance_sampling_ratio/max": 1.5931972448642437, "eval_entropy": 2.1048372708834133, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4151367247104645, "eval_reward_meter_mean": 0.4592640560406905, "eval_reward_meter_std": 0.3293467943484967, "eval_reward_count_adherence_mean": 0.9569021555093619, "eval_reward_count_adherence_std": 0.07193636994522351, "eval_reward_arabic_clean_mean": 0.9423076923076923, "eval_reward_arabic_clean_std": 0.12560976010102493, "eval_reward_total_composite_mean": 0.4151367247104645, "eval_reward_total_composite_std": 0.3096239452178662, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 200.0} {"timestamp_utc": "2026-04-11T19:46:40Z", "mode": "train", "global_step": 201, "epoch": 0.007761816496756255, "loss": 0.0034, "grad_norm": 12.455371856689453, "learning_rate": 9.393939393939396e-06, "num_tokens": 431437.0, "completions/mean_length": 31.875, "completions/min_length": 29.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9721503257751465, "rewards/meter/std": 0.02047812007367611, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9721503257751465, "rewards/total_composite/std": 0.02047812007367611, "reward": 0.9721503257751465, "reward_std": 0.02047811821103096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13416853547096252, "sampling/sampling_logp_difference/max": 1.1965785026550293, "sampling/importance_sampling_ratio/min": 0.3022264838218689, "sampling/importance_sampling_ratio/mean": 1.007079839706421, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9935045540332794, "clip_ratio/low_mean": 0.0364175159484148, "clip_ratio/low_min": 0.0364175159484148, "clip_ratio/high_mean": 0.08195106312632561, "clip_ratio/high_max": 0.08195106312632561, "clip_ratio/region_mean": 0.11836857907474041, "reward_total_mean": 0.9721503257751465, "reward_meter_mean": 0.9721503257751465, "reward_meter_std": 0.02047812007367611, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9721503257751465, "reward_total_composite_std": 0.02047812007367611, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 201.0} {"timestamp_utc": "2026-04-11T19:46:50Z", "mode": "train", "global_step": 202, "epoch": 0.0078004324992276795, "loss": -0.126, "grad_norm": 2.5677430629730225, "learning_rate": 9.390909090909092e-06, "num_tokens": 436473.0, "completions/mean_length": 443.5, "completions/min_length": 371.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 433.71429443359375, "completions/min_terminated_length": 371.0, "completions/max_terminated_length": 494.0, "rewards/meter/mean": 0.4077759385108948, "rewards/meter/std": 0.2577785849571228, "rewards/count_adherence/mean": 0.9318181872367859, "rewards/count_adherence/std": 0.06428243219852448, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3806021213531494, "rewards/total_composite/std": 0.23827770352363586, "reward": 0.3806021213531494, "reward_std": 0.23827768862247467, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1868906170129776, "sampling/sampling_logp_difference/max": 1.5403366088867188, "sampling/importance_sampling_ratio/min": 0.2143089473247528, "sampling/importance_sampling_ratio/mean": 1.0397894382476807, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.633182942867279, "clip_ratio/low_mean": 0.046742528676986694, "clip_ratio/low_min": 0.046742528676986694, "clip_ratio/high_mean": 0.05716947093605995, "clip_ratio/high_max": 0.05716947093605995, "clip_ratio/region_mean": 0.10391199961304665, "reward_total_mean": 0.3806021213531494, "reward_meter_mean": 0.4077759385108948, "reward_meter_std": 0.2577785849571228, "reward_count_adherence_mean": 0.9318181872367859, "reward_count_adherence_std": 0.06428243219852448, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3806021213531494, "reward_total_composite_std": 0.23827770352363586, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 202.0} {"timestamp_utc": "2026-04-11T19:46:56Z", "mode": "train", "global_step": 203, "epoch": 0.007839048501699104, "loss": 0.0063, "grad_norm": 10.374756813049316, "learning_rate": 9.387878787878789e-06, "num_tokens": 438144.0, "completions/mean_length": 54.875, "completions/min_length": 44.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.875, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.18955467641353607, "rewards/meter/std": 0.2376922219991684, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.18955467641353607, "rewards/total_composite/std": 0.2376922219991684, "reward": 0.18955467641353607, "reward_std": 0.2376922219991684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19148828089237213, "sampling/sampling_logp_difference/max": 1.4236383438110352, "sampling/importance_sampling_ratio/min": 0.24083620309829712, "sampling/importance_sampling_ratio/mean": 1.0263407230377197, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.969245657324791, "clip_ratio/low_mean": 0.13778485730290413, "clip_ratio/low_min": 0.13778485730290413, "clip_ratio/high_mean": 0.05425724573433399, "clip_ratio/high_max": 0.05425724573433399, "clip_ratio/region_mean": 0.19204210303723812, "reward_total_mean": 0.18955467641353607, "reward_meter_mean": 0.18955467641353607, "reward_meter_std": 0.2376922219991684, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.18955467641353607, "reward_total_composite_std": 0.2376922219991684, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 203.0} {"timestamp_utc": "2026-04-11T19:47:04Z", "mode": "train", "global_step": 204, "epoch": 0.007877664504170528, "loss": 0.0447, "grad_norm": 3.7568657398223877, "learning_rate": 9.384848484848486e-06, "num_tokens": 442326.0, "completions/mean_length": 316.75, "completions/min_length": 280.0, "completions/max_length": 341.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 316.75, "completions/min_terminated_length": 280.0, "completions/max_terminated_length": 341.0, "rewards/meter/mean": 0.7115911245346069, "rewards/meter/std": 0.30422648787498474, "rewards/count_adherence/mean": 0.9722222089767456, "rewards/count_adherence/std": 0.05143444985151291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6847348213195801, "rewards/total_composite/std": 0.28163081407546997, "reward": 0.6847348213195801, "reward_std": 0.28163081407546997, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14612703025341034, "sampling/sampling_logp_difference/max": 2.316192150115967, "sampling/importance_sampling_ratio/min": 0.09864851087331772, "sampling/importance_sampling_ratio/mean": 1.0315098762512207, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8388848304748535, "clip_ratio/low_mean": 0.05318944435566664, "clip_ratio/low_min": 0.05318944435566664, "clip_ratio/high_mean": 0.05784035939723253, "clip_ratio/high_max": 0.05784035939723253, "clip_ratio/region_mean": 0.11102980375289917, "reward_total_mean": 0.6847348213195801, "reward_meter_mean": 0.7115911245346069, "reward_meter_std": 0.30422648787498474, "reward_count_adherence_mean": 0.9722222089767456, "reward_count_adherence_std": 0.05143444985151291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6847348213195801, "reward_total_composite_std": 0.28163081407546997, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 204.0} {"timestamp_utc": "2026-04-11T19:47:08Z", "mode": "train", "global_step": 205, "epoch": 0.007916280506641952, "loss": -0.0141, "grad_norm": 22.554645538330078, "learning_rate": 9.381818181818183e-06, "num_tokens": 443612.0, "completions/mean_length": 20.75, "completions/min_length": 19.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.75, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.2569815218448639, "rewards/meter/std": 0.17644450068473816, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.2569815218448639, "rewards/total_composite/std": 0.17644450068473816, "reward": 0.2569815218448639, "reward_std": 0.17644448578357697, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11848119646310806, "sampling/sampling_logp_difference/max": 2.3156137466430664, "sampling/importance_sampling_ratio/min": 0.09870558977127075, "sampling/importance_sampling_ratio/mean": 1.010235071182251, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47789650596678257, "clip_ratio/low_mean": 0.04515550099313259, "clip_ratio/low_min": 0.04515550099313259, "clip_ratio/high_mean": 0.04085497930645943, "clip_ratio/high_max": 0.04085497930645943, "clip_ratio/region_mean": 0.08601048029959202, "reward_total_mean": 0.2569815218448639, "reward_meter_mean": 0.2569815218448639, "reward_meter_std": 0.17644450068473816, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.2569815218448639, "reward_total_composite_std": 0.17644450068473816, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 205.0} {"timestamp_utc": "2026-04-11T19:47:18Z", "mode": "train", "global_step": 206, "epoch": 0.007954896509113376, "loss": -0.2172, "grad_norm": 2.1156201362609863, "learning_rate": 9.378787878787879e-06, "num_tokens": 448589.0, "completions/mean_length": 459.125, "completions/min_length": 410.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 451.5714416503906, "completions/min_terminated_length": 410.0, "completions/max_terminated_length": 502.0, "rewards/meter/mean": 0.5943694114685059, "rewards/meter/std": 0.28580430150032043, "rewards/count_adherence/mean": 0.8928571343421936, "rewards/count_adherence/std": 0.05399491637945175, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.542489767074585, "rewards/total_composite/std": 0.26988792419433594, "reward": 0.542489767074585, "reward_std": 0.2698879539966583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16989928483963013, "sampling/sampling_logp_difference/max": 1.390742301940918, "sampling/importance_sampling_ratio/min": 0.2488904893398285, "sampling/importance_sampling_ratio/mean": 1.0428776741027832, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3715617954730988, "clip_ratio/low_mean": 0.02254154160618782, "clip_ratio/low_min": 0.02254154160618782, "clip_ratio/high_mean": 0.08172544464468956, "clip_ratio/high_max": 0.08172544464468956, "clip_ratio/region_mean": 0.10426698625087738, "reward_total_mean": 0.542489767074585, "reward_meter_mean": 0.5943694114685059, "reward_meter_std": 0.28580430150032043, "reward_count_adherence_mean": 0.8928571343421936, "reward_count_adherence_std": 0.05399491637945175, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.542489767074585, "reward_total_composite_std": 0.26988792419433594, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 206.0} {"timestamp_utc": "2026-04-11T19:47:25Z", "mode": "train", "global_step": 207, "epoch": 0.0079935125115848, "loss": 0.0262, "grad_norm": 5.197041988372803, "learning_rate": 9.375757575757576e-06, "num_tokens": 451666.0, "completions/mean_length": 200.625, "completions/min_length": 185.0, "completions/max_length": 223.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 200.625, "completions/min_terminated_length": 185.0, "completions/max_terminated_length": 223.0, "rewards/meter/mean": 0.5436156988143921, "rewards/meter/std": 0.36589959263801575, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5297455787658691, "rewards/total_composite/std": 0.3861840069293976, "reward": 0.5297455787658691, "reward_std": 0.3861840069293976, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17878292500972748, "sampling/sampling_logp_difference/max": 1.3839550018310547, "sampling/importance_sampling_ratio/min": 0.25058552622795105, "sampling/importance_sampling_ratio/mean": 1.0335437059402466, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5693641006946564, "clip_ratio/low_mean": 0.03460816852748394, "clip_ratio/low_min": 0.03460816852748394, "clip_ratio/high_mean": 0.09698447678238153, "clip_ratio/high_max": 0.09698447678238153, "clip_ratio/region_mean": 0.13159264530986547, "reward_total_mean": 0.5297455787658691, "reward_meter_mean": 0.5436156988143921, "reward_meter_std": 0.36589959263801575, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5297455787658691, "reward_total_composite_std": 0.3861840069293976, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 207.0} {"timestamp_utc": "2026-04-11T19:47:34Z", "mode": "train", "global_step": 208, "epoch": 0.008032128514056224, "loss": -0.0625, "grad_norm": 4.907017707824707, "learning_rate": 9.372727272727273e-06, "num_tokens": 453242.0, "completions/mean_length": 112.0, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 54.857147216796875, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.5209001302719116, "rewards/meter/std": 0.4826778173446655, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5157449841499329, "rewards/total_composite/std": 0.48821765184402466, "reward": 0.5157449841499329, "reward_std": 0.48821765184402466, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1978602260351181, "sampling/sampling_logp_difference/max": 1.7859477996826172, "sampling/importance_sampling_ratio/min": 0.16763810813426971, "sampling/importance_sampling_ratio/mean": 1.0225262641906738, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.829095020890236, "clip_ratio/low_mean": 0.04633831046521664, "clip_ratio/low_min": 0.04633831046521664, "clip_ratio/high_mean": 0.08172481320798397, "clip_ratio/high_max": 0.08172481320798397, "clip_ratio/region_mean": 0.1280631236732006, "reward_total_mean": 0.5157449841499329, "reward_meter_mean": 0.5209001302719116, "reward_meter_std": 0.4826778173446655, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5157449841499329, "reward_total_composite_std": 0.48821765184402466, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 208.0} {"timestamp_utc": "2026-04-11T19:47:39Z", "mode": "train", "global_step": 209, "epoch": 0.008070744516527648, "loss": 0.0425, "grad_norm": 10.775023460388184, "learning_rate": 9.36969696969697e-06, "num_tokens": 455075.0, "completions/mean_length": 65.125, "completions/min_length": 60.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.687164843082428, "rewards/meter/std": 0.42298591136932373, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.687164843082428, "rewards/total_composite/std": 0.42298591136932373, "reward": 0.687164843082428, "reward_std": 0.42298588156700134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1489512026309967, "sampling/sampling_logp_difference/max": 1.3404273986816406, "sampling/importance_sampling_ratio/min": 0.2617337703704834, "sampling/importance_sampling_ratio/mean": 1.031755805015564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5745535343885422, "clip_ratio/low_mean": 0.05209819972515106, "clip_ratio/low_min": 0.05209819972515106, "clip_ratio/high_mean": 0.07205317448824644, "clip_ratio/high_max": 0.07205317448824644, "clip_ratio/region_mean": 0.1241513742133975, "reward_total_mean": 0.687164843082428, "reward_meter_mean": 0.687164843082428, "reward_meter_std": 0.42298591136932373, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.687164843082428, "reward_total_composite_std": 0.42298591136932373, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 209.0} {"timestamp_utc": "2026-04-11T19:47:44Z", "mode": "train", "global_step": 210, "epoch": 0.008109360518999072, "loss": 0.0918, "grad_norm": 8.813285827636719, "learning_rate": 9.366666666666668e-06, "num_tokens": 457201.0, "completions/mean_length": 94.75, "completions/min_length": 74.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.75, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.8014340400695801, "rewards/meter/std": 0.26059165596961975, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8014340400695801, "rewards/total_composite/std": 0.26059165596961975, "reward": 0.8014340400695801, "reward_std": 0.26059165596961975, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16893059015274048, "sampling/sampling_logp_difference/max": 1.4793734550476074, "sampling/importance_sampling_ratio/min": 0.22778037190437317, "sampling/importance_sampling_ratio/mean": 1.0492236614227295, "sampling/importance_sampling_ratio/max": 1.846545934677124, "entropy": 2.1869092285633087, "clip_ratio/low_mean": 0.06012810207903385, "clip_ratio/low_min": 0.06012810207903385, "clip_ratio/high_mean": 0.0914016654714942, "clip_ratio/high_max": 0.0914016654714942, "clip_ratio/region_mean": 0.15152976755052805, "reward_total_mean": 0.8014340400695801, "reward_meter_mean": 0.8014340400695801, "reward_meter_std": 0.26059165596961975, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8014340400695801, "reward_total_composite_std": 0.26059165596961975, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 210.0} {"timestamp_utc": "2026-04-11T19:47:48Z", "mode": "train", "global_step": 211, "epoch": 0.008147976521470498, "loss": 0.043, "grad_norm": 14.590749740600586, "learning_rate": 9.363636363636365e-06, "num_tokens": 458609.0, "completions/mean_length": 31.0, "completions/min_length": 26.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.3326827883720398, "rewards/meter/std": 0.391814649105072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3326827883720398, "rewards/total_composite/std": 0.391814649105072, "reward": 0.3326827883720398, "reward_std": 0.39181461930274963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17955826222896576, "sampling/sampling_logp_difference/max": 1.105809211730957, "sampling/importance_sampling_ratio/min": 0.3309429883956909, "sampling/importance_sampling_ratio/mean": 1.0280555486679077, "sampling/importance_sampling_ratio/max": 1.9337586164474487, "entropy": 2.1823814064264297, "clip_ratio/low_mean": 0.0616428735665977, "clip_ratio/low_min": 0.0616428735665977, "clip_ratio/high_mean": 0.06869588885456324, "clip_ratio/high_max": 0.06869588885456324, "clip_ratio/region_mean": 0.13033876242116094, "reward_total_mean": 0.3326827883720398, "reward_meter_mean": 0.3326827883720398, "reward_meter_std": 0.391814649105072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3326827883720398, "reward_total_composite_std": 0.391814649105072, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 211.0} {"timestamp_utc": "2026-04-11T19:47:52Z", "mode": "train", "global_step": 212, "epoch": 0.008186592523941922, "loss": 0.1396, "grad_norm": 20.575647354125977, "learning_rate": 9.36060606060606e-06, "num_tokens": 460114.0, "completions/mean_length": 35.125, "completions/min_length": 26.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.125, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.45719897747039795, "rewards/meter/std": 0.3513868451118469, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.45719897747039795, "rewards/total_composite/std": 0.3513868451118469, "reward": 0.45719897747039795, "reward_std": 0.35138681530952454, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18254390358924866, "sampling/sampling_logp_difference/max": 1.7746210098266602, "sampling/importance_sampling_ratio/min": 0.1695476919412613, "sampling/importance_sampling_ratio/mean": 0.9953309297561646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3654158115386963, "clip_ratio/low_mean": 0.10219099884852767, "clip_ratio/low_min": 0.10219099884852767, "clip_ratio/high_mean": 0.0721238311380148, "clip_ratio/high_max": 0.0721238311380148, "clip_ratio/region_mean": 0.17431482998654246, "reward_total_mean": 0.45719897747039795, "reward_meter_mean": 0.45719897747039795, "reward_meter_std": 0.3513868451118469, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.45719897747039795, "reward_total_composite_std": 0.3513868451118469, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 212.0} {"timestamp_utc": "2026-04-11T19:47:57Z", "mode": "train", "global_step": 213, "epoch": 0.008225208526413346, "loss": 0.0285, "grad_norm": 7.170069694519043, "learning_rate": 9.357575757575758e-06, "num_tokens": 462364.0, "completions/mean_length": 106.25, "completions/min_length": 85.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.25, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.6657600998878479, "rewards/meter/std": 0.4287368059158325, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6657600998878479, "rewards/total_composite/std": 0.4287368059158325, "reward": 0.6657600998878479, "reward_std": 0.4287368059158325, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16524526476860046, "sampling/sampling_logp_difference/max": 1.5607881546020508, "sampling/importance_sampling_ratio/min": 0.20997050404548645, "sampling/importance_sampling_ratio/mean": 1.0314472913742065, "sampling/importance_sampling_ratio/max": 1.96308434009552, "entropy": 2.2049818336963654, "clip_ratio/low_mean": 0.039724151603877544, "clip_ratio/low_min": 0.039724151603877544, "clip_ratio/high_mean": 0.09881766140460968, "clip_ratio/high_max": 0.09881766140460968, "clip_ratio/region_mean": 0.13854181300848722, "reward_total_mean": 0.6657600998878479, "reward_meter_mean": 0.6657600998878479, "reward_meter_std": 0.4287368059158325, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6657600998878479, "reward_total_composite_std": 0.4287368059158325, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 213.0} {"timestamp_utc": "2026-04-11T19:48:02Z", "mode": "train", "global_step": 214, "epoch": 0.00826382452888477, "loss": 0.0144, "grad_norm": 12.966662406921387, "learning_rate": 9.354545454545455e-06, "num_tokens": 463910.0, "completions/mean_length": 37.25, "completions/min_length": 29.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9338778257369995, "rewards/meter/std": 0.12218131124973297, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9338778257369995, "rewards/total_composite/std": 0.12218131124973297, "reward": 0.9338778257369995, "reward_std": 0.12218130379915237, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15117491781711578, "sampling/sampling_logp_difference/max": 1.2583513259887695, "sampling/importance_sampling_ratio/min": 0.2841220498085022, "sampling/importance_sampling_ratio/mean": 1.0329978466033936, "sampling/importance_sampling_ratio/max": 1.8531534671783447, "entropy": 1.585478514432907, "clip_ratio/low_mean": 0.02658250369131565, "clip_ratio/low_min": 0.02658250369131565, "clip_ratio/high_mean": 0.08606619853526354, "clip_ratio/high_max": 0.08606619853526354, "clip_ratio/region_mean": 0.11264870222657919, "reward_total_mean": 0.9338778257369995, "reward_meter_mean": 0.9338778257369995, "reward_meter_std": 0.12218131124973297, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9338778257369995, "reward_total_composite_std": 0.12218131124973297, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 214.0} {"timestamp_utc": "2026-04-11T19:48:06Z", "mode": "train", "global_step": 215, "epoch": 0.008302440531356195, "loss": 0.239, "grad_norm": 16.243061065673828, "learning_rate": 9.351515151515152e-06, "num_tokens": 465457.0, "completions/mean_length": 34.375, "completions/min_length": 27.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.375, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.6347050666809082, "rewards/meter/std": 0.485779345035553, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6347050666809082, "rewards/total_composite/std": 0.485779345035553, "reward": 0.6347050666809082, "reward_std": 0.485779345035553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17325210571289062, "sampling/sampling_logp_difference/max": 1.1446723937988281, "sampling/importance_sampling_ratio/min": 0.31832820177078247, "sampling/importance_sampling_ratio/mean": 1.0188474655151367, "sampling/importance_sampling_ratio/max": 1.7199817895889282, "entropy": 1.831810086965561, "clip_ratio/low_mean": 0.04079106356948614, "clip_ratio/low_min": 0.04079106356948614, "clip_ratio/high_mean": 0.09631683025509119, "clip_ratio/high_max": 0.09631683025509119, "clip_ratio/region_mean": 0.13710789382457733, "reward_total_mean": 0.6347050666809082, "reward_meter_mean": 0.6347050666809082, "reward_meter_std": 0.485779345035553, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6347050666809082, "reward_total_composite_std": 0.485779345035553, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 215.0} {"timestamp_utc": "2026-04-11T19:48:17Z", "mode": "train", "global_step": 216, "epoch": 0.008341056533827619, "loss": -0.0905, "grad_norm": 0.7487542033195496, "learning_rate": 9.34848484848485e-06, "num_tokens": 467753.0, "completions/mean_length": 507.0, "completions/min_length": 472.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.875, "completions/mean_terminated_length": 472.0, "completions/min_terminated_length": 472.0, "completions/max_terminated_length": 472.0, "rewards/meter/mean": 0.3023744523525238, "rewards/meter/std": 0.17562586069107056, "rewards/count_adherence/mean": 0.637499988079071, "rewards/count_adherence/std": 0.06408698856830597, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.18635782599449158, "rewards/total_composite/std": 0.1041322648525238, "reward": 0.18635782599449158, "reward_std": 0.1041322648525238, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18534618616104126, "sampling/sampling_logp_difference/max": 1.2733802795410156, "sampling/importance_sampling_ratio/min": 0.2798839509487152, "sampling/importance_sampling_ratio/mean": 1.041240930557251, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36599498987197876, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.015360169112682343, "clip_ratio/high_max": 0.015360169112682343, "clip_ratio/region_mean": 0.015360169112682343, "reward_total_mean": 0.18635782599449158, "reward_meter_mean": 0.3023744523525238, "reward_meter_std": 0.17562586069107056, "reward_count_adherence_mean": 0.637499988079071, "reward_count_adherence_std": 0.06408698856830597, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.18635782599449158, "reward_total_composite_std": 0.1041322648525238, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 216.0} {"timestamp_utc": "2026-04-11T19:48:21Z", "mode": "train", "global_step": 217, "epoch": 0.008379672536299043, "loss": -0.0148, "grad_norm": 11.847907066345215, "learning_rate": 9.345454545454547e-06, "num_tokens": 469457.0, "completions/mean_length": 52.0, "completions/min_length": 45.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.5340137481689453, "rewards/meter/std": 0.3664604425430298, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5340137481689453, "rewards/total_composite/std": 0.3664604425430298, "reward": 0.5340137481689453, "reward_std": 0.3664604127407074, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19996923208236694, "sampling/sampling_logp_difference/max": 1.6229486465454102, "sampling/importance_sampling_ratio/min": 0.1973160207271576, "sampling/importance_sampling_ratio/mean": 1.0360443592071533, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0648878514766693, "clip_ratio/low_mean": 0.08761653117835522, "clip_ratio/low_min": 0.08761653117835522, "clip_ratio/high_mean": 0.0961015336215496, "clip_ratio/high_max": 0.0961015336215496, "clip_ratio/region_mean": 0.18371806479990482, "reward_total_mean": 0.5340137481689453, "reward_meter_mean": 0.5340137481689453, "reward_meter_std": 0.3664604425430298, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5340137481689453, "reward_total_composite_std": 0.3664604425430298, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 217.0} {"timestamp_utc": "2026-04-11T19:48:27Z", "mode": "train", "global_step": 218, "epoch": 0.008418288538770467, "loss": -0.0238, "grad_norm": 8.575998306274414, "learning_rate": 9.342424242424243e-06, "num_tokens": 471653.0, "completions/mean_length": 105.5, "completions/min_length": 97.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.5, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.5959634184837341, "rewards/meter/std": 0.32533523440361023, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5959634184837341, "rewards/total_composite/std": 0.32533523440361023, "reward": 0.5959634184837341, "reward_std": 0.32533523440361023, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16784748435020447, "sampling/sampling_logp_difference/max": 1.3933839797973633, "sampling/importance_sampling_ratio/min": 0.2482338696718216, "sampling/importance_sampling_ratio/mean": 1.0328844785690308, "sampling/importance_sampling_ratio/max": 1.9590723514556885, "entropy": 1.9123822152614594, "clip_ratio/low_mean": 0.05961098615080118, "clip_ratio/low_min": 0.05961098615080118, "clip_ratio/high_mean": 0.07770076114684343, "clip_ratio/high_max": 0.07770076114684343, "clip_ratio/region_mean": 0.13731174729764462, "reward_total_mean": 0.5959634184837341, "reward_meter_mean": 0.5959634184837341, "reward_meter_std": 0.32533523440361023, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5959634184837341, "reward_total_composite_std": 0.32533523440361023, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 218.0} {"timestamp_utc": "2026-04-11T19:48:32Z", "mode": "train", "global_step": 219, "epoch": 0.008456904541241891, "loss": 0.0253, "grad_norm": 7.932456016540527, "learning_rate": 9.33939393939394e-06, "num_tokens": 474198.0, "completions/mean_length": 132.125, "completions/min_length": 125.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.125, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.4994851350784302, "rewards/meter/std": 0.3480455279350281, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4994851350784302, "rewards/total_composite/std": 0.3480455279350281, "reward": 0.4994851350784302, "reward_std": 0.3480455279350281, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17217673361301422, "sampling/sampling_logp_difference/max": 1.7459349632263184, "sampling/importance_sampling_ratio/min": 0.17448177933692932, "sampling/importance_sampling_ratio/mean": 1.0123792886734009, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5948337465524673, "clip_ratio/low_mean": 0.07042216509580612, "clip_ratio/low_min": 0.07042216509580612, "clip_ratio/high_mean": 0.08742227591574192, "clip_ratio/high_max": 0.08742227591574192, "clip_ratio/region_mean": 0.15784444101154804, "reward_total_mean": 0.4994851350784302, "reward_meter_mean": 0.4994851350784302, "reward_meter_std": 0.3480455279350281, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4994851350784302, "reward_total_composite_std": 0.3480455279350281, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 219.0} {"timestamp_utc": "2026-04-11T19:48:37Z", "mode": "train", "global_step": 220, "epoch": 0.008495520543713315, "loss": -0.0252, "grad_norm": 9.201763153076172, "learning_rate": 9.336363636363637e-06, "num_tokens": 475969.0, "completions/mean_length": 66.375, "completions/min_length": 59.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.5444612503051758, "rewards/meter/std": 0.46134647727012634, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5444612503051758, "rewards/total_composite/std": 0.46134647727012634, "reward": 0.5444612503051758, "reward_std": 0.46134647727012634, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15713398158550262, "sampling/sampling_logp_difference/max": 1.1669750213623047, "sampling/importance_sampling_ratio/min": 0.31130722165107727, "sampling/importance_sampling_ratio/mean": 1.0116194486618042, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6451235264539719, "clip_ratio/low_mean": 0.06260535214096308, "clip_ratio/low_min": 0.06260535214096308, "clip_ratio/high_mean": 0.06848039291799068, "clip_ratio/high_max": 0.06848039291799068, "clip_ratio/region_mean": 0.13108574505895376, "reward_total_mean": 0.5444612503051758, "reward_meter_mean": 0.5444612503051758, "reward_meter_std": 0.46134647727012634, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5444612503051758, "reward_total_composite_std": 0.46134647727012634, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 220.0} {"timestamp_utc": "2026-04-11T19:48:42Z", "mode": "train", "global_step": 221, "epoch": 0.00853413654618474, "loss": -0.0093, "grad_norm": 11.238251686096191, "learning_rate": 9.333333333333334e-06, "num_tokens": 477742.0, "completions/mean_length": 58.625, "completions/min_length": 41.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.5600961446762085, "rewards/meter/std": 0.4695318341255188, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5600961446762085, "rewards/total_composite/std": 0.4695318341255188, "reward": 0.5600961446762085, "reward_std": 0.4695318341255188, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16859595477581024, "sampling/sampling_logp_difference/max": 1.9469690322875977, "sampling/importance_sampling_ratio/min": 0.14270596206188202, "sampling/importance_sampling_ratio/mean": 1.021318793296814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5837604850530624, "clip_ratio/low_mean": 0.07524601556360722, "clip_ratio/low_min": 0.07524601556360722, "clip_ratio/high_mean": 0.07579075917601585, "clip_ratio/high_max": 0.07579075917601585, "clip_ratio/region_mean": 0.15103677473962307, "reward_total_mean": 0.5600961446762085, "reward_meter_mean": 0.5600961446762085, "reward_meter_std": 0.4695318341255188, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5600961446762085, "reward_total_composite_std": 0.4695318341255188, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 221.0} {"timestamp_utc": "2026-04-11T19:48:46Z", "mode": "train", "global_step": 222, "epoch": 0.008572752548656163, "loss": -0.147, "grad_norm": 11.750017166137695, "learning_rate": 9.33030303030303e-06, "num_tokens": 479254.0, "completions/mean_length": 36.0, "completions/min_length": 22.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.8671466112136841, "rewards/meter/std": 0.3500947654247284, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8671466112136841, "rewards/total_composite/std": 0.3500947654247284, "reward": 0.8671466112136841, "reward_std": 0.3500947654247284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16731053590774536, "sampling/sampling_logp_difference/max": 1.9990720748901367, "sampling/importance_sampling_ratio/min": 0.13546092808246613, "sampling/importance_sampling_ratio/mean": 1.0165225267410278, "sampling/importance_sampling_ratio/max": 1.9629331827163696, "entropy": 1.5899460166692734, "clip_ratio/low_mean": 0.022727273404598236, "clip_ratio/low_min": 0.022727273404598236, "clip_ratio/high_mean": 0.09998656460084021, "clip_ratio/high_max": 0.09998656460084021, "clip_ratio/region_mean": 0.12271383800543845, "reward_total_mean": 0.8671466112136841, "reward_meter_mean": 0.8671466112136841, "reward_meter_std": 0.3500947654247284, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8671466112136841, "reward_total_composite_std": 0.3500947654247284, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 222.0} {"timestamp_utc": "2026-04-11T19:48:51Z", "mode": "train", "global_step": 223, "epoch": 0.008611368551127587, "loss": -0.0678, "grad_norm": 7.995184421539307, "learning_rate": 9.327272727272729e-06, "num_tokens": 481092.0, "completions/mean_length": 73.75, "completions/min_length": 55.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.75, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.8009564280509949, "rewards/meter/std": 0.26855042576789856, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8009564280509949, "rewards/total_composite/std": 0.26855042576789856, "reward": 0.8009564280509949, "reward_std": 0.26855039596557617, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17098720371723175, "sampling/sampling_logp_difference/max": 1.2840356826782227, "sampling/importance_sampling_ratio/min": 0.2769174873828888, "sampling/importance_sampling_ratio/mean": 1.0155538320541382, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9959132969379425, "clip_ratio/low_mean": 0.05262349545955658, "clip_ratio/low_min": 0.05262349545955658, "clip_ratio/high_mean": 0.07255261763930321, "clip_ratio/high_max": 0.07255261763930321, "clip_ratio/region_mean": 0.1251761130988598, "reward_total_mean": 0.8009564280509949, "reward_meter_mean": 0.8009564280509949, "reward_meter_std": 0.26855042576789856, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8009564280509949, "reward_total_composite_std": 0.26855042576789856, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 223.0} {"timestamp_utc": "2026-04-11T19:48:56Z", "mode": "train", "global_step": 224, "epoch": 0.008649984553599012, "loss": 0.0128, "grad_norm": 8.910178184509277, "learning_rate": 9.324242424242424e-06, "num_tokens": 483221.0, "completions/mean_length": 102.125, "completions/min_length": 98.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.125, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.7856420278549194, "rewards/meter/std": 0.2833492159843445, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7856420278549194, "rewards/total_composite/std": 0.2833492159843445, "reward": 0.7856420278549194, "reward_std": 0.2833492159843445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16320450603961945, "sampling/sampling_logp_difference/max": 1.6196975708007812, "sampling/importance_sampling_ratio/min": 0.1979585438966751, "sampling/importance_sampling_ratio/mean": 1.028478741645813, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7342596799135208, "clip_ratio/low_mean": 0.04744089301675558, "clip_ratio/low_min": 0.04744089301675558, "clip_ratio/high_mean": 0.10716481134295464, "clip_ratio/high_max": 0.10716481134295464, "clip_ratio/region_mean": 0.15460570435971022, "reward_total_mean": 0.7856420278549194, "reward_meter_mean": 0.7856420278549194, "reward_meter_std": 0.2833492159843445, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7856420278549194, "reward_total_composite_std": 0.2833492159843445, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 224.0} {"timestamp_utc": "2026-04-11T19:49:02Z", "mode": "train", "global_step": 225, "epoch": 0.008688600556070436, "loss": 0.0627, "grad_norm": 5.88389253616333, "learning_rate": 9.321212121212122e-06, "num_tokens": 486360.0, "completions/mean_length": 190.375, "completions/min_length": 153.0, "completions/max_length": 211.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 190.375, "completions/min_terminated_length": 153.0, "completions/max_terminated_length": 211.0, "rewards/meter/mean": 0.680176854133606, "rewards/meter/std": 0.38275107741355896, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625820279121399, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6395675539970398, "rewards/total_composite/std": 0.3551376760005951, "reward": 0.6395675539970398, "reward_std": 0.3551376461982727, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16371257603168488, "sampling/sampling_logp_difference/max": 2.192625045776367, "sampling/importance_sampling_ratio/min": 0.1116233542561531, "sampling/importance_sampling_ratio/mean": 1.033612608909607, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.218223586678505, "clip_ratio/low_mean": 0.04819977842271328, "clip_ratio/low_min": 0.04819977842271328, "clip_ratio/high_mean": 0.0867786854505539, "clip_ratio/high_max": 0.0867786854505539, "clip_ratio/region_mean": 0.13497846387326717, "reward_total_mean": 0.6395675539970398, "reward_meter_mean": 0.680176854133606, "reward_meter_std": 0.38275107741355896, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625820279121399, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6395675539970398, "reward_total_composite_std": 0.3551376760005951, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 225.0} {"timestamp_utc": "2026-04-11T19:49:08Z", "mode": "train", "global_step": 226, "epoch": 0.00872721655854186, "loss": 0.0006, "grad_norm": 4.9813737869262695, "learning_rate": 9.318181818181819e-06, "num_tokens": 489110.0, "completions/mean_length": 157.75, "completions/min_length": 140.0, "completions/max_length": 170.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.75, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.8852719068527222, "rewards/meter/std": 0.17902569472789764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8852719068527222, "rewards/total_composite/std": 0.17902569472789764, "reward": 0.8852719068527222, "reward_std": 0.17902567982673645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14676453173160553, "sampling/sampling_logp_difference/max": 1.2427654266357422, "sampling/importance_sampling_ratio/min": 0.2885850667953491, "sampling/importance_sampling_ratio/mean": 1.0297491550445557, "sampling/importance_sampling_ratio/max": 1.984347939491272, "entropy": 1.9846401810646057, "clip_ratio/low_mean": 0.030824373476207256, "clip_ratio/low_min": 0.030824373476207256, "clip_ratio/high_mean": 0.09920443315058947, "clip_ratio/high_max": 0.09920443315058947, "clip_ratio/region_mean": 0.13002880662679672, "reward_total_mean": 0.8852719068527222, "reward_meter_mean": 0.8852719068527222, "reward_meter_std": 0.17902569472789764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8852719068527222, "reward_total_composite_std": 0.17902569472789764, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 226.0} {"timestamp_utc": "2026-04-11T19:49:12Z", "mode": "train", "global_step": 227, "epoch": 0.008765832561013284, "loss": 0.1164, "grad_norm": 10.270444869995117, "learning_rate": 9.315151515151516e-06, "num_tokens": 490872.0, "completions/mean_length": 56.25, "completions/min_length": 37.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.36732152104377747, "rewards/meter/std": 0.30656546354293823, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.36732152104377747, "rewards/total_composite/std": 0.30656546354293823, "reward": 0.36732152104377747, "reward_std": 0.30656546354293823, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18055380880832672, "sampling/sampling_logp_difference/max": 1.549325942993164, "sampling/importance_sampling_ratio/min": 0.21239109337329865, "sampling/importance_sampling_ratio/mean": 1.0084258317947388, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.041958436369896, "clip_ratio/low_mean": 0.07808315567672253, "clip_ratio/low_min": 0.07808315567672253, "clip_ratio/high_mean": 0.09229664131999016, "clip_ratio/high_max": 0.09229664131999016, "clip_ratio/region_mean": 0.17037979699671268, "reward_total_mean": 0.36732152104377747, "reward_meter_mean": 0.36732152104377747, "reward_meter_std": 0.30656546354293823, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.36732152104377747, "reward_total_composite_std": 0.30656546354293823, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 227.0} {"timestamp_utc": "2026-04-11T19:49:17Z", "mode": "train", "global_step": 228, "epoch": 0.008804448563484708, "loss": 0.0772, "grad_norm": 9.097184181213379, "learning_rate": 9.312121212121212e-06, "num_tokens": 492930.0, "completions/mean_length": 91.25, "completions/min_length": 82.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.25, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.8796967267990112, "rewards/meter/std": 0.2009752243757248, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8796967267990112, "rewards/total_composite/std": 0.2009752243757248, "reward": 0.8796967267990112, "reward_std": 0.2009752094745636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14313152432441711, "sampling/sampling_logp_difference/max": 1.881901741027832, "sampling/importance_sampling_ratio/min": 0.15230019390583038, "sampling/importance_sampling_ratio/mean": 1.02587890625, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5294092446565628, "clip_ratio/low_mean": 0.027772237546741962, "clip_ratio/low_min": 0.027772237546741962, "clip_ratio/high_mean": 0.1098766503855586, "clip_ratio/high_max": 0.1098766503855586, "clip_ratio/region_mean": 0.13764888793230057, "reward_total_mean": 0.8796967267990112, "reward_meter_mean": 0.8796967267990112, "reward_meter_std": 0.2009752243757248, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8796967267990112, "reward_total_composite_std": 0.2009752243757248, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 228.0} {"timestamp_utc": "2026-04-11T19:49:24Z", "mode": "train", "global_step": 229, "epoch": 0.008843064565956132, "loss": 0.0304, "grad_norm": 4.891687393188477, "learning_rate": 9.30909090909091e-06, "num_tokens": 496362.0, "completions/mean_length": 215.0, "completions/min_length": 199.0, "completions/max_length": 232.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 215.0, "completions/min_terminated_length": 199.0, "completions/max_terminated_length": 232.0, "rewards/meter/mean": 0.5281928777694702, "rewards/meter/std": 0.36326658725738525, "rewards/count_adherence/mean": 0.910714328289032, "rewards/count_adherence/std": 0.07393559068441391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4802461862564087, "rewards/total_composite/std": 0.3348220884799957, "reward": 0.4802461862564087, "reward_std": 0.3348220884799957, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14938952028751373, "sampling/sampling_logp_difference/max": 1.3926191329956055, "sampling/importance_sampling_ratio/min": 0.24842379987239838, "sampling/importance_sampling_ratio/mean": 1.028792381286621, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8009164333343506, "clip_ratio/low_mean": 0.0690256292000413, "clip_ratio/low_min": 0.0690256292000413, "clip_ratio/high_mean": 0.04163628350943327, "clip_ratio/high_max": 0.04163628350943327, "clip_ratio/region_mean": 0.11066191270947456, "reward_total_mean": 0.4802461862564087, "reward_meter_mean": 0.5281928777694702, "reward_meter_std": 0.36326658725738525, "reward_count_adherence_mean": 0.910714328289032, "reward_count_adherence_std": 0.07393559068441391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4802461862564087, "reward_total_composite_std": 0.3348220884799957, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 229.0} {"timestamp_utc": "2026-04-11T19:49:29Z", "mode": "train", "global_step": 230, "epoch": 0.008881680568427556, "loss": -0.0263, "grad_norm": 8.536412239074707, "learning_rate": 9.306060606060608e-06, "num_tokens": 498671.0, "completions/mean_length": 117.625, "completions/min_length": 83.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.625, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.723436713218689, "rewards/meter/std": 0.3901357650756836, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.723436713218689, "rewards/total_composite/std": 0.3901357650756836, "reward": 0.723436713218689, "reward_std": 0.3901357650756836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17858588695526123, "sampling/sampling_logp_difference/max": 1.1610007286071777, "sampling/importance_sampling_ratio/min": 0.3272934556007385, "sampling/importance_sampling_ratio/mean": 1.0375686883926392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9786228984594345, "clip_ratio/low_mean": 0.03780912980437279, "clip_ratio/low_min": 0.03780912980437279, "clip_ratio/high_mean": 0.12172973342239857, "clip_ratio/high_max": 0.12172973342239857, "clip_ratio/region_mean": 0.15953886322677135, "reward_total_mean": 0.723436713218689, "reward_meter_mean": 0.723436713218689, "reward_meter_std": 0.3901357650756836, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.723436713218689, "reward_total_composite_std": 0.3901357650756836, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 230.0} {"timestamp_utc": "2026-04-11T19:49:34Z", "mode": "train", "global_step": 231, "epoch": 0.00892029657089898, "loss": 0.0366, "grad_norm": 6.538661003112793, "learning_rate": 9.303030303030303e-06, "num_tokens": 500877.0, "completions/mean_length": 115.75, "completions/min_length": 106.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.75, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.6978538036346436, "rewards/meter/std": 0.3173922300338745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6978538036346436, "rewards/total_composite/std": 0.3173922300338745, "reward": 0.6978538036346436, "reward_std": 0.3173922002315521, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15994611382484436, "sampling/sampling_logp_difference/max": 1.3233470916748047, "sampling/importance_sampling_ratio/min": 0.2662426829338074, "sampling/importance_sampling_ratio/mean": 1.036413550376892, "sampling/importance_sampling_ratio/max": 1.9082386493682861, "entropy": 2.1323368698358536, "clip_ratio/low_mean": 0.06935460586100817, "clip_ratio/low_min": 0.06935460586100817, "clip_ratio/high_mean": 0.07807724550366402, "clip_ratio/high_max": 0.07807724550366402, "clip_ratio/region_mean": 0.14743185136467218, "reward_total_mean": 0.6978538036346436, "reward_meter_mean": 0.6978538036346436, "reward_meter_std": 0.3173922300338745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6978538036346436, "reward_total_composite_std": 0.3173922300338745, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 231.0} {"timestamp_utc": "2026-04-11T19:49:40Z", "mode": "train", "global_step": 232, "epoch": 0.008958912573370404, "loss": 0.0737, "grad_norm": 6.832438945770264, "learning_rate": 9.3e-06, "num_tokens": 503508.0, "completions/mean_length": 147.875, "completions/min_length": 122.0, "completions/max_length": 168.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.875, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 168.0, "rewards/meter/mean": 0.7667824029922485, "rewards/meter/std": 0.29820093512535095, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7425558567047119, "rewards/total_composite/std": 0.2870854437351227, "reward": 0.7425558567047119, "reward_std": 0.2870854139328003, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1671251356601715, "sampling/sampling_logp_difference/max": 1.5834159851074219, "sampling/importance_sampling_ratio/min": 0.20527270436286926, "sampling/importance_sampling_ratio/mean": 1.031022071838379, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.93074631690979, "clip_ratio/low_mean": 0.03656993992626667, "clip_ratio/low_min": 0.03656993992626667, "clip_ratio/high_mean": 0.12983744032680988, "clip_ratio/high_max": 0.12983744032680988, "clip_ratio/region_mean": 0.16640738025307655, "reward_total_mean": 0.7425558567047119, "reward_meter_mean": 0.7667824029922485, "reward_meter_std": 0.29820093512535095, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7425558567047119, "reward_total_composite_std": 0.2870854437351227, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 232.0} {"timestamp_utc": "2026-04-11T19:49:50Z", "mode": "train", "global_step": 233, "epoch": 0.008997528575841829, "loss": -0.0188, "grad_norm": 1.5289084911346436, "learning_rate": 9.296969696969698e-06, "num_tokens": 506784.0, "completions/mean_length": 502.5, "completions/min_length": 480.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 486.66668701171875, "completions/min_terminated_length": 480.0, "completions/max_terminated_length": 497.0, "rewards/meter/mean": 0.6105531454086304, "rewards/meter/std": 0.27458783984184265, "rewards/count_adherence/mean": 0.53125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3117174506187439, "rewards/total_composite/std": 0.19874346256256104, "reward": 0.3117174506187439, "reward_std": 0.19874344766139984, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12929628789424896, "sampling/sampling_logp_difference/max": 1.6299529075622559, "sampling/importance_sampling_ratio/min": 0.19593879580497742, "sampling/importance_sampling_ratio/mean": 1.0306029319763184, "sampling/importance_sampling_ratio/max": 1.9305408000946045, "entropy": 0.6105214804410934, "clip_ratio/low_mean": 0.009834368713200092, "clip_ratio/low_min": 0.009834368713200092, "clip_ratio/high_mean": 0.023773369379341602, "clip_ratio/high_max": 0.023773369379341602, "clip_ratio/region_mean": 0.033607738092541695, "reward_total_mean": 0.3117174506187439, "reward_meter_mean": 0.6105531454086304, "reward_meter_std": 0.27458783984184265, "reward_count_adherence_mean": 0.53125, "reward_count_adherence_std": 0.25877460837364197, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3117174506187439, "reward_total_composite_std": 0.19874346256256104, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 233.0} {"timestamp_utc": "2026-04-11T19:49:55Z", "mode": "train", "global_step": 234, "epoch": 0.009036144578313253, "loss": 0.1226, "grad_norm": 17.917465209960938, "learning_rate": 9.293939393939395e-06, "num_tokens": 508394.0, "completions/mean_length": 42.25, "completions/min_length": 33.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.25, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.44727417826652527, "rewards/meter/std": 0.3157028555870056, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.44727417826652527, "rewards/total_composite/std": 0.3157028555870056, "reward": 0.44727417826652527, "reward_std": 0.3157028555870056, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20457275211811066, "sampling/sampling_logp_difference/max": 2.2530784606933594, "sampling/importance_sampling_ratio/min": 0.10507524758577347, "sampling/importance_sampling_ratio/mean": 0.9895703792572021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.387149639427662, "clip_ratio/low_mean": 0.09165013581514359, "clip_ratio/low_min": 0.09165013581514359, "clip_ratio/high_mean": 0.09212539158761501, "clip_ratio/high_max": 0.09212539158761501, "clip_ratio/region_mean": 0.1837755274027586, "reward_total_mean": 0.44727417826652527, "reward_meter_mean": 0.44727417826652527, "reward_meter_std": 0.3157028555870056, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.44727417826652527, "reward_total_composite_std": 0.3157028555870056, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 234.0} {"timestamp_utc": "2026-04-11T19:49:59Z", "mode": "train", "global_step": 235, "epoch": 0.009074760580784677, "loss": 0.034, "grad_norm": 9.29736042022705, "learning_rate": 9.29090909090909e-06, "num_tokens": 510345.0, "completions/mean_length": 73.875, "completions/min_length": 69.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.875, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9192884564399719, "rewards/meter/std": 0.1364532858133316, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9192884564399719, "rewards/total_composite/std": 0.1364532858133316, "reward": 0.9192884564399719, "reward_std": 0.1364532709121704, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15465568006038666, "sampling/sampling_logp_difference/max": 1.8014373779296875, "sampling/importance_sampling_ratio/min": 0.16506147384643555, "sampling/importance_sampling_ratio/mean": 1.0462172031402588, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.776310458779335, "clip_ratio/low_mean": 0.02009246125817299, "clip_ratio/low_min": 0.02009246125817299, "clip_ratio/high_mean": 0.10756326653063297, "clip_ratio/high_max": 0.10756326653063297, "clip_ratio/region_mean": 0.12765572778880596, "reward_total_mean": 0.9192884564399719, "reward_meter_mean": 0.9192884564399719, "reward_meter_std": 0.1364532858133316, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9192884564399719, "reward_total_composite_std": 0.1364532858133316, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 235.0} {"timestamp_utc": "2026-04-11T19:50:04Z", "mode": "train", "global_step": 236, "epoch": 0.0091133765832561, "loss": 0.01, "grad_norm": 8.880298614501953, "learning_rate": 9.28787878787879e-06, "num_tokens": 512262.0, "completions/mean_length": 75.625, "completions/min_length": 67.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.6458913087844849, "rewards/meter/std": 0.33680447936058044, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6458913087844849, "rewards/total_composite/std": 0.33680447936058044, "reward": 0.6458913087844849, "reward_std": 0.33680450916290283, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14490646123886108, "sampling/sampling_logp_difference/max": 1.312744140625, "sampling/importance_sampling_ratio/min": 0.26908063888549805, "sampling/importance_sampling_ratio/mean": 1.0316869020462036, "sampling/importance_sampling_ratio/max": 1.9332317113876343, "entropy": 1.796240046620369, "clip_ratio/low_mean": 0.06175778992474079, "clip_ratio/low_min": 0.06175778992474079, "clip_ratio/high_mean": 0.07200214825570583, "clip_ratio/high_max": 0.07200214825570583, "clip_ratio/region_mean": 0.13375993818044662, "reward_total_mean": 0.6458913087844849, "reward_meter_mean": 0.6458913087844849, "reward_meter_std": 0.33680447936058044, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6458913087844849, "reward_total_composite_std": 0.33680447936058044, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 236.0} {"timestamp_utc": "2026-04-11T19:50:09Z", "mode": "train", "global_step": 237, "epoch": 0.009151992585727525, "loss": 0.0414, "grad_norm": 8.00853443145752, "learning_rate": 9.284848484848485e-06, "num_tokens": 514491.0, "completions/mean_length": 95.625, "completions/min_length": 87.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.625, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.4600679576396942, "rewards/meter/std": 0.4352971911430359, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4599551856517792, "rewards/total_composite/std": 0.43543270230293274, "reward": 0.4599551856517792, "reward_std": 0.43543270230293274, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1568225622177124, "sampling/sampling_logp_difference/max": 1.0943379402160645, "sampling/importance_sampling_ratio/min": 0.3347611427307129, "sampling/importance_sampling_ratio/mean": 1.032072901725769, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7081756442785263, "clip_ratio/low_mean": 0.08504617214202881, "clip_ratio/low_min": 0.08504617214202881, "clip_ratio/high_mean": 0.05680421367287636, "clip_ratio/high_max": 0.05680421367287636, "clip_ratio/region_mean": 0.14185038581490517, "reward_total_mean": 0.4599551856517792, "reward_meter_mean": 0.4600679576396942, "reward_meter_std": 0.4352971911430359, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4599551856517792, "reward_total_composite_std": 0.43543270230293274, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 237.0} {"timestamp_utc": "2026-04-11T19:50:16Z", "mode": "train", "global_step": 238, "epoch": 0.009190608588198949, "loss": 0.1335, "grad_norm": 5.644252300262451, "learning_rate": 9.281818181818183e-06, "num_tokens": 517253.0, "completions/mean_length": 166.25, "completions/min_length": 128.0, "completions/max_length": 221.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.25, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 221.0, "rewards/meter/mean": 0.7441527843475342, "rewards/meter/std": 0.38449952006340027, "rewards/count_adherence/mean": 0.7250000238418579, "rewards/count_adherence/std": 0.14880476891994476, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5001842379570007, "rewards/total_composite/std": 0.37060803174972534, "reward": 0.5001842379570007, "reward_std": 0.37060803174972534, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16780243813991547, "sampling/sampling_logp_difference/max": 1.4439067840576172, "sampling/importance_sampling_ratio/min": 0.23600395023822784, "sampling/importance_sampling_ratio/mean": 1.0297116041183472, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2700495421886444, "clip_ratio/low_mean": 0.03305934276431799, "clip_ratio/low_min": 0.03305934276431799, "clip_ratio/high_mean": 0.09207058418542147, "clip_ratio/high_max": 0.09207058418542147, "clip_ratio/region_mean": 0.12512992694973946, "reward_total_mean": 0.5001842379570007, "reward_meter_mean": 0.7441527843475342, "reward_meter_std": 0.38449952006340027, "reward_count_adherence_mean": 0.7250000238418579, "reward_count_adherence_std": 0.14880476891994476, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5001842379570007, "reward_total_composite_std": 0.37060803174972534, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 238.0} {"timestamp_utc": "2026-04-11T19:50:22Z", "mode": "train", "global_step": 239, "epoch": 0.009229224590670373, "loss": 0.0538, "grad_norm": 6.0654401779174805, "learning_rate": 9.27878787878788e-06, "num_tokens": 519716.0, "completions/mean_length": 118.875, "completions/min_length": 106.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.875, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.5814238786697388, "rewards/meter/std": 0.36550426483154297, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5465434193611145, "rewards/total_composite/std": 0.35062775015830994, "reward": 0.5465434193611145, "reward_std": 0.35062772035598755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17544740438461304, "sampling/sampling_logp_difference/max": 1.1976795196533203, "sampling/importance_sampling_ratio/min": 0.301893949508667, "sampling/importance_sampling_ratio/mean": 1.043744683265686, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.404376655817032, "clip_ratio/low_mean": 0.06852707732468843, "clip_ratio/low_min": 0.06852707732468843, "clip_ratio/high_mean": 0.07056730799376965, "clip_ratio/high_max": 0.07056730799376965, "clip_ratio/region_mean": 0.13909438531845808, "reward_total_mean": 0.5465434193611145, "reward_meter_mean": 0.5814238786697388, "reward_meter_std": 0.36550426483154297, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5465434193611145, "reward_total_composite_std": 0.35062775015830994, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 239.0} {"timestamp_utc": "2026-04-11T19:50:27Z", "mode": "train", "global_step": 240, "epoch": 0.009267840593141797, "loss": 0.0204, "grad_norm": 8.986166000366211, "learning_rate": 9.275757575757577e-06, "num_tokens": 521863.0, "completions/mean_length": 101.375, "completions/min_length": 77.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.375, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.9709064960479736, "rewards/meter/std": 0.027551084756851196, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9709064960479736, "rewards/total_composite/std": 0.027551084756851196, "reward": 0.9709064960479736, "reward_std": 0.02755107544362545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16349507868289948, "sampling/sampling_logp_difference/max": 1.1616079807281494, "sampling/importance_sampling_ratio/min": 0.3129824995994568, "sampling/importance_sampling_ratio/mean": 1.0282772779464722, "sampling/importance_sampling_ratio/max": 1.8251866102218628, "entropy": 2.066037192940712, "clip_ratio/low_mean": 0.04993662517517805, "clip_ratio/low_min": 0.04993662517517805, "clip_ratio/high_mean": 0.0725616067647934, "clip_ratio/high_max": 0.0725616067647934, "clip_ratio/region_mean": 0.12249823193997145, "reward_total_mean": 0.9709064960479736, "reward_meter_mean": 0.9709064960479736, "reward_meter_std": 0.027551084756851196, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9709064960479736, "reward_total_composite_std": 0.027551084756851196, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 240.0} {"timestamp_utc": "2026-04-11T19:50:32Z", "mode": "train", "global_step": 241, "epoch": 0.009306456595613221, "loss": 0.0892, "grad_norm": 8.806304931640625, "learning_rate": 9.272727272727273e-06, "num_tokens": 523847.0, "completions/mean_length": 85.0, "completions/min_length": 70.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.45606791973114014, "rewards/meter/std": 0.38195329904556274, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.2626701593399048, "rewards/total_composite/std": 0.22439196705818176, "reward": 0.2626701593399048, "reward_std": 0.22439199686050415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1751883327960968, "sampling/sampling_logp_difference/max": 1.5793533325195312, "sampling/importance_sampling_ratio/min": 0.20610834658145905, "sampling/importance_sampling_ratio/mean": 1.031845211982727, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1117777675390244, "clip_ratio/low_mean": 0.0722261555492878, "clip_ratio/low_min": 0.0722261555492878, "clip_ratio/high_mean": 0.06964285857975483, "clip_ratio/high_max": 0.06964285857975483, "clip_ratio/region_mean": 0.14186901412904263, "reward_total_mean": 0.2626701593399048, "reward_meter_mean": 0.45606791973114014, "reward_meter_std": 0.38195329904556274, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.2626701593399048, "reward_total_composite_std": 0.22439196705818176, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 241.0} {"timestamp_utc": "2026-04-11T19:50:37Z", "mode": "train", "global_step": 242, "epoch": 0.009345072598084645, "loss": 0.0767, "grad_norm": 8.190821647644043, "learning_rate": 9.26969696969697e-06, "num_tokens": 525777.0, "completions/mean_length": 83.25, "completions/min_length": 74.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.25, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.8624944686889648, "rewards/meter/std": 0.33260369300842285, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.800595223903656, "rewards/total_composite/std": 0.3509736657142639, "reward": 0.800595223903656, "reward_std": 0.35097363591194153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16949887573719025, "sampling/sampling_logp_difference/max": 1.3167352676391602, "sampling/importance_sampling_ratio/min": 0.26800885796546936, "sampling/importance_sampling_ratio/mean": 1.0372215509414673, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2592698484659195, "clip_ratio/low_mean": 0.03575989790260792, "clip_ratio/low_min": 0.03575989790260792, "clip_ratio/high_mean": 0.11791071016341448, "clip_ratio/high_max": 0.11791071016341448, "clip_ratio/region_mean": 0.1536706080660224, "reward_total_mean": 0.800595223903656, "reward_meter_mean": 0.8624944686889648, "reward_meter_std": 0.33260369300842285, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.800595223903656, "reward_total_composite_std": 0.3509736657142639, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 242.0} {"timestamp_utc": "2026-04-11T19:50:42Z", "mode": "train", "global_step": 243, "epoch": 0.009383688600556071, "loss": 0.0226, "grad_norm": 9.102118492126465, "learning_rate": 9.266666666666667e-06, "num_tokens": 527584.0, "completions/mean_length": 62.875, "completions/min_length": 57.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.875, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.969333827495575, "rewards/meter/std": 0.05763925239443779, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.969333827495575, "rewards/total_composite/std": 0.05763925239443779, "reward": 0.969333827495575, "reward_std": 0.0576392337679863, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15184138715267181, "sampling/sampling_logp_difference/max": 1.2962350845336914, "sampling/importance_sampling_ratio/min": 0.2735597789287567, "sampling/importance_sampling_ratio/mean": 1.0088082551956177, "sampling/importance_sampling_ratio/max": 1.6970453262329102, "entropy": 1.626624509692192, "clip_ratio/low_mean": 0.01953125, "clip_ratio/low_min": 0.01953125, "clip_ratio/high_mean": 0.14678451232612133, "clip_ratio/high_max": 0.14678451232612133, "clip_ratio/region_mean": 0.16631576232612133, "reward_total_mean": 0.969333827495575, "reward_meter_mean": 0.969333827495575, "reward_meter_std": 0.05763925239443779, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.969333827495575, "reward_total_composite_std": 0.05763925239443779, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 243.0} {"timestamp_utc": "2026-04-11T19:50:47Z", "mode": "train", "global_step": 244, "epoch": 0.009422304603027495, "loss": 0.0658, "grad_norm": 7.276552677154541, "learning_rate": 9.263636363636364e-06, "num_tokens": 529780.0, "completions/mean_length": 118.5, "completions/min_length": 72.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.5, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.3871627748012543, "rewards/meter/std": 0.4140600562095642, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.356499046087265, "rewards/total_composite/std": 0.3705804646015167, "reward": 0.356499046087265, "reward_std": 0.3705804646015167, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17408481240272522, "sampling/sampling_logp_difference/max": 1.224247932434082, "sampling/importance_sampling_ratio/min": 0.2939786911010742, "sampling/importance_sampling_ratio/mean": 1.0257729291915894, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.060980185866356, "clip_ratio/low_mean": 0.07911759708076715, "clip_ratio/low_min": 0.07911759708076715, "clip_ratio/high_mean": 0.074860580265522, "clip_ratio/high_max": 0.074860580265522, "clip_ratio/region_mean": 0.15397817734628916, "reward_total_mean": 0.356499046087265, "reward_meter_mean": 0.3871627748012543, "reward_meter_std": 0.4140600562095642, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.356499046087265, "reward_total_composite_std": 0.3705804646015167, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 244.0} {"timestamp_utc": "2026-04-11T19:50:52Z", "mode": "train", "global_step": 245, "epoch": 0.00946092060549892, "loss": -0.024, "grad_norm": 7.2777557373046875, "learning_rate": 9.260606060606062e-06, "num_tokens": 532179.0, "completions/mean_length": 122.875, "completions/min_length": 101.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.875, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.6986033320426941, "rewards/meter/std": 0.25926199555397034, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6617265939712524, "rewards/total_composite/std": 0.27683618664741516, "reward": 0.6617265939712524, "reward_std": 0.2768361568450928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15021459758281708, "sampling/sampling_logp_difference/max": 1.3864390850067139, "sampling/importance_sampling_ratio/min": 0.29009345173835754, "sampling/importance_sampling_ratio/mean": 1.0323891639709473, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7008240818977356, "clip_ratio/low_mean": 0.07482307218015194, "clip_ratio/low_min": 0.07482307218015194, "clip_ratio/high_mean": 0.05074426345527172, "clip_ratio/high_max": 0.05074426345527172, "clip_ratio/region_mean": 0.12556733563542366, "reward_total_mean": 0.6617265939712524, "reward_meter_mean": 0.6986033320426941, "reward_meter_std": 0.25926199555397034, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6617265939712524, "reward_total_composite_std": 0.27683618664741516, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 245.0} {"timestamp_utc": "2026-04-11T19:51:02Z", "mode": "train", "global_step": 246, "epoch": 0.009499536607970344, "loss": 0.0104, "grad_norm": 4.300860404968262, "learning_rate": 9.257575757575759e-06, "num_tokens": 535079.0, "completions/mean_length": 208.5, "completions/min_length": 147.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 165.1428680419922, "completions/min_terminated_length": 147.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.622548520565033, "rewards/meter/std": 0.3121766746044159, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.16690459847450256, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.39897841215133667, "rewards/total_composite/std": 0.22350633144378662, "reward": 0.39897841215133667, "reward_std": 0.22350633144378662, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1846722960472107, "sampling/sampling_logp_difference/max": 1.6206417083740234, "sampling/importance_sampling_ratio/min": 0.19777174293994904, "sampling/importance_sampling_ratio/mean": 1.0414752960205078, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3770622611045837, "clip_ratio/low_mean": 0.05107419705018401, "clip_ratio/low_min": 0.05107419705018401, "clip_ratio/high_mean": 0.05021729413419962, "clip_ratio/high_max": 0.05021729413419962, "clip_ratio/region_mean": 0.10129149118438363, "reward_total_mean": 0.39897841215133667, "reward_meter_mean": 0.622548520565033, "reward_meter_std": 0.3121766746044159, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.16690459847450256, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.39897841215133667, "reward_total_composite_std": 0.22350633144378662, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 246.0} {"timestamp_utc": "2026-04-11T19:51:07Z", "mode": "train", "global_step": 247, "epoch": 0.009538152610441768, "loss": 0.1449, "grad_norm": 9.519201278686523, "learning_rate": 9.254545454545454e-06, "num_tokens": 536963.0, "completions/mean_length": 73.5, "completions/min_length": 55.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.5, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.7730816006660461, "rewards/meter/std": 0.314135879278183, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7730816006660461, "rewards/total_composite/std": 0.314135879278183, "reward": 0.7730816006660461, "reward_std": 0.314135879278183, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18582503497600555, "sampling/sampling_logp_difference/max": 1.5778687000274658, "sampling/importance_sampling_ratio/min": 0.2123212218284607, "sampling/importance_sampling_ratio/mean": 1.046431541442871, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.306705191731453, "clip_ratio/low_mean": 0.017326733097434044, "clip_ratio/low_min": 0.017326733097434044, "clip_ratio/high_mean": 0.13099952787160873, "clip_ratio/high_max": 0.13099952787160873, "clip_ratio/region_mean": 0.14832626096904278, "reward_total_mean": 0.7730816006660461, "reward_meter_mean": 0.7730816006660461, "reward_meter_std": 0.314135879278183, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7730816006660461, "reward_total_composite_std": 0.314135879278183, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 247.0} {"timestamp_utc": "2026-04-11T19:51:13Z", "mode": "train", "global_step": 248, "epoch": 0.009576768612913192, "loss": -0.0386, "grad_norm": 4.550097942352295, "learning_rate": 9.251515151515152e-06, "num_tokens": 540434.0, "completions/mean_length": 224.875, "completions/min_length": 187.0, "completions/max_length": 244.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 224.875, "completions/min_terminated_length": 187.0, "completions/max_terminated_length": 244.0, "rewards/meter/mean": 0.9458253979682922, "rewards/meter/std": 0.07403269410133362, "rewards/count_adherence/mean": 0.828125, "rewards/count_adherence/std": 0.09300298243761063, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.782259464263916, "rewards/total_composite/std": 0.10182006657123566, "reward": 0.782259464263916, "reward_std": 0.10182006657123566, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15468242764472961, "sampling/sampling_logp_difference/max": 1.4476947784423828, "sampling/importance_sampling_ratio/min": 0.23511166870594025, "sampling/importance_sampling_ratio/mean": 1.0300755500793457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9894641190767288, "clip_ratio/low_mean": 0.051101832650601864, "clip_ratio/low_min": 0.051101832650601864, "clip_ratio/high_mean": 0.08469811640679836, "clip_ratio/high_max": 0.08469811640679836, "clip_ratio/region_mean": 0.13579994905740023, "reward_total_mean": 0.782259464263916, "reward_meter_mean": 0.9458253979682922, "reward_meter_std": 0.07403269410133362, "reward_count_adherence_mean": 0.828125, "reward_count_adherence_std": 0.09300298243761063, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.782259464263916, "reward_total_composite_std": 0.10182006657123566, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 248.0} {"timestamp_utc": "2026-04-11T19:51:18Z", "mode": "train", "global_step": 249, "epoch": 0.009615384615384616, "loss": 0.0622, "grad_norm": 10.006518363952637, "learning_rate": 9.248484848484849e-06, "num_tokens": 542207.0, "completions/mean_length": 72.625, "completions/min_length": 62.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.625, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.3977659046649933, "rewards/meter/std": 0.394779771566391, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.39560720324516296, "rewards/total_composite/std": 0.3970901072025299, "reward": 0.39560720324516296, "reward_std": 0.3970900774002075, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20861367881298065, "sampling/sampling_logp_difference/max": 1.7951688766479492, "sampling/importance_sampling_ratio/min": 0.1660993993282318, "sampling/importance_sampling_ratio/mean": 1.0430378913879395, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.432593807578087, "clip_ratio/low_mean": 0.08824052847921848, "clip_ratio/low_min": 0.08824052847921848, "clip_ratio/high_mean": 0.05096696317195892, "clip_ratio/high_max": 0.05096696317195892, "clip_ratio/region_mean": 0.1392074916511774, "reward_total_mean": 0.39560720324516296, "reward_meter_mean": 0.3977659046649933, "reward_meter_std": 0.394779771566391, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.39560720324516296, "reward_total_composite_std": 0.3970901072025299, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 249.0} {"timestamp_utc": "2026-04-11T19:51:22Z", "mode": "train", "global_step": 250, "epoch": 0.00965400061785604, "loss": 0.0013, "grad_norm": 13.965093612670898, "learning_rate": 9.245454545454546e-06, "num_tokens": 543733.0, "completions/mean_length": 34.75, "completions/min_length": 32.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7543537616729736, "rewards/meter/std": 0.32807299494743347, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7543537616729736, "rewards/total_composite/std": 0.32807299494743347, "reward": 0.7543537616729736, "reward_std": 0.32807299494743347, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17254644632339478, "sampling/sampling_logp_difference/max": 1.6727294921875, "sampling/importance_sampling_ratio/min": 0.2496013343334198, "sampling/importance_sampling_ratio/mean": 0.9995890259742737, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6636070609092712, "clip_ratio/low_mean": 0.03012663428671658, "clip_ratio/low_min": 0.03012663428671658, "clip_ratio/high_mean": 0.1064140466041863, "clip_ratio/high_max": 0.1064140466041863, "clip_ratio/region_mean": 0.13654068089090288, "reward_total_mean": 0.7543537616729736, "reward_meter_mean": 0.7543537616729736, "reward_meter_std": 0.32807299494743347, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7543537616729736, "reward_total_composite_std": 0.32807299494743347, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 250.0} {"timestamp_utc": "2026-04-11T19:52:53Z", "mode": "eval", "global_step": 250, "epoch": 0.00965400061785604, "eval_loss": NaN, "eval_runtime": 90.6589, "eval_samples_per_second": 1.147, "eval_steps_per_second": 0.143, "eval_num_tokens": 543733.0, "eval_completions/mean_length": 228.9903846153846, "eval_completions/min_length": 53.38461538461539, "eval_completions/max_length": 494.7692307692308, "eval_completions/clipped_ratio": 0.17307692307692307, "eval_completions/mean_terminated_length": 168.23443838266226, "eval_completions/min_terminated_length": 53.38461538461539, "eval_completions/max_terminated_length": 335.0, "eval_rewards/meter/mean": 0.47416638411008394, "eval_rewards/meter/std": 0.3508666604757309, "eval_rewards/count_adherence/mean": 0.7803410750169021, "eval_rewards/count_adherence/std": 0.24469699080173785, "eval_rewards/arabic_clean/mean": 0.8942307692307693, "eval_rewards/arabic_clean/std": 0.21981406670350295, "eval_rewards/total_composite/mean": 0.3687195135996892, "eval_rewards/total_composite/std": 0.32943402574612546, "eval_reward": 0.3687195135996892, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.12459980008693841, "eval_sampling/sampling_logp_difference/max": 1.2276372909545898, "eval_sampling/importance_sampling_ratio/min": 0.29775738372252536, "eval_sampling/importance_sampling_ratio/mean": 1.034668812384972, "eval_sampling/importance_sampling_ratio/max": 1.5637650306408222, "eval_entropy": 1.889658808708191, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.3687195135996892, "eval_reward_meter_mean": 0.47416638411008394, "eval_reward_meter_std": 0.3508666604757309, "eval_reward_count_adherence_mean": 0.7803410750169021, "eval_reward_count_adherence_std": 0.24469699080173785, "eval_reward_arabic_clean_mean": 0.8942307692307693, "eval_reward_arabic_clean_std": 0.21981406670350295, "eval_reward_total_composite_mean": 0.3687195135996892, "eval_reward_total_composite_std": 0.32943402574612546, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 250.0} {"timestamp_utc": "2026-04-11T19:53:00Z", "mode": "train", "global_step": 251, "epoch": 0.009692616620327464, "loss": 0.0119, "grad_norm": 11.79161262512207, "learning_rate": 9.242424242424244e-06, "num_tokens": 545690.0, "completions/mean_length": 65.625, "completions/min_length": 55.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.625, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.6017528176307678, "rewards/meter/std": 0.44121459126472473, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5393182635307312, "rewards/total_composite/std": 0.41130462288856506, "reward": 0.5393182635307312, "reward_std": 0.41130462288856506, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1789507418870926, "sampling/sampling_logp_difference/max": 1.51678466796875, "sampling/importance_sampling_ratio/min": 0.21941626071929932, "sampling/importance_sampling_ratio/mean": 1.021461844444275, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0140078961849213, "clip_ratio/low_mean": 0.04646642506122589, "clip_ratio/low_min": 0.04646642506122589, "clip_ratio/high_mean": 0.10757232829928398, "clip_ratio/high_max": 0.10757232829928398, "clip_ratio/region_mean": 0.15403875336050987, "reward_total_mean": 0.5393182635307312, "reward_meter_mean": 0.6017528176307678, "reward_meter_std": 0.44121459126472473, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5393182635307312, "reward_total_composite_std": 0.41130462288856506, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 251.0} {"timestamp_utc": "2026-04-11T19:53:05Z", "mode": "train", "global_step": 252, "epoch": 0.009731232622798888, "loss": 0.0093, "grad_norm": 6.874846935272217, "learning_rate": 9.23939393939394e-06, "num_tokens": 547982.0, "completions/mean_length": 109.5, "completions/min_length": 89.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.5, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.7877767086029053, "rewards/meter/std": 0.299540638923645, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6217153668403625, "rewards/total_composite/std": 0.2621327042579651, "reward": 0.6217153668403625, "reward_std": 0.2621327042579651, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1741628497838974, "sampling/sampling_logp_difference/max": 1.4004230499267578, "sampling/importance_sampling_ratio/min": 0.2464926689863205, "sampling/importance_sampling_ratio/mean": 1.0337904691696167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3438350409269333, "clip_ratio/low_mean": 0.05587371252477169, "clip_ratio/low_min": 0.05587371252477169, "clip_ratio/high_mean": 0.08756711520254612, "clip_ratio/high_max": 0.08756711520254612, "clip_ratio/region_mean": 0.1434408277273178, "reward_total_mean": 0.6217153668403625, "reward_meter_mean": 0.7877767086029053, "reward_meter_std": 0.299540638923645, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6217153668403625, "reward_total_composite_std": 0.2621327042579651, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 252.0} {"timestamp_utc": "2026-04-11T19:53:10Z", "mode": "train", "global_step": 253, "epoch": 0.009769848625270312, "loss": 0.0319, "grad_norm": 14.97449016571045, "learning_rate": 9.236363636363636e-06, "num_tokens": 549313.0, "completions/mean_length": 34.375, "completions/min_length": 28.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.375, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.98116135597229, "rewards/meter/std": 0.015678538009524345, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.98116135597229, "rewards/total_composite/std": 0.015678538009524345, "reward": 0.98116135597229, "reward_std": 0.01567855291068554, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17273853719234467, "sampling/sampling_logp_difference/max": 1.310826301574707, "sampling/importance_sampling_ratio/min": 0.2695971727371216, "sampling/importance_sampling_ratio/mean": 1.0338517427444458, "sampling/importance_sampling_ratio/max": 1.7639540433883667, "entropy": 2.204130381345749, "clip_ratio/low_mean": 0.0667356732301414, "clip_ratio/low_min": 0.0667356732301414, "clip_ratio/high_mean": 0.07423552963882685, "clip_ratio/high_max": 0.07423552963882685, "clip_ratio/region_mean": 0.14097120286896825, "reward_total_mean": 0.98116135597229, "reward_meter_mean": 0.98116135597229, "reward_meter_std": 0.015678538009524345, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.98116135597229, "reward_total_composite_std": 0.015678538009524345, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 253.0} {"timestamp_utc": "2026-04-11T19:53:14Z", "mode": "train", "global_step": 254, "epoch": 0.009808464627741736, "loss": 0.0434, "grad_norm": 8.44826602935791, "learning_rate": 9.233333333333334e-06, "num_tokens": 551128.0, "completions/mean_length": 77.875, "completions/min_length": 69.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.875, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.8029962182044983, "rewards/meter/std": 0.34782952070236206, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8029962182044983, "rewards/total_composite/std": 0.34782952070236206, "reward": 0.8029962182044983, "reward_std": 0.34782955050468445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15848758816719055, "sampling/sampling_logp_difference/max": 0.89579176902771, "sampling/importance_sampling_ratio/min": 0.41547152400016785, "sampling/importance_sampling_ratio/mean": 1.0570210218429565, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8999166786670685, "clip_ratio/low_mean": 0.036308735609054565, "clip_ratio/low_min": 0.036308735609054565, "clip_ratio/high_mean": 0.09618495963513851, "clip_ratio/high_max": 0.09618495963513851, "clip_ratio/region_mean": 0.13249369524419308, "reward_total_mean": 0.8029962182044983, "reward_meter_mean": 0.8029962182044983, "reward_meter_std": 0.34782952070236206, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8029962182044983, "reward_total_composite_std": 0.34782952070236206, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 254.0} {"timestamp_utc": "2026-04-11T19:53:24Z", "mode": "train", "global_step": 255, "epoch": 0.00984708063021316, "loss": 0.0437, "grad_norm": 9.855134963989258, "learning_rate": 9.23030303030303e-06, "num_tokens": 552746.0, "completions/mean_length": 98.25, "completions/min_length": 22.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 39.142860412597656, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.8700155019760132, "rewards/meter/std": 0.2133215367794037, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8700155019760132, "rewards/total_composite/std": 0.2133215367794037, "reward": 0.8700155019760132, "reward_std": 0.2133215218782425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17322853207588196, "sampling/sampling_logp_difference/max": 1.207193374633789, "sampling/importance_sampling_ratio/min": 0.29903537034988403, "sampling/importance_sampling_ratio/mean": 1.0299566984176636, "sampling/importance_sampling_ratio/max": 1.803825855255127, "entropy": 2.0371719151735306, "clip_ratio/low_mean": 0.03721843124367297, "clip_ratio/low_min": 0.03721843124367297, "clip_ratio/high_mean": 0.0770185561850667, "clip_ratio/high_max": 0.0770185561850667, "clip_ratio/region_mean": 0.11423698742873967, "reward_total_mean": 0.8700155019760132, "reward_meter_mean": 0.8700155019760132, "reward_meter_std": 0.2133215367794037, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8700155019760132, "reward_total_composite_std": 0.2133215367794037, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 255.0} {"timestamp_utc": "2026-04-11T19:53:32Z", "mode": "train", "global_step": 256, "epoch": 0.009885696632684585, "loss": 0.169, "grad_norm": 4.932265758514404, "learning_rate": 9.227272727272728e-06, "num_tokens": 555578.0, "completions/mean_length": 162.0, "completions/min_length": 106.0, "completions/max_length": 365.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 162.0, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 365.0, "rewards/meter/mean": 0.7288707494735718, "rewards/meter/std": 0.32933905720710754, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.24800792336463928, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6166548728942871, "rewards/total_composite/std": 0.34062808752059937, "reward": 0.6166548728942871, "reward_std": 0.34062808752059937, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1348879486322403, "sampling/sampling_logp_difference/max": 1.468480110168457, "sampling/importance_sampling_ratio/min": 0.2302752137184143, "sampling/importance_sampling_ratio/mean": 1.0188031196594238, "sampling/importance_sampling_ratio/max": 1.7278470993041992, "entropy": 1.8008338958024979, "clip_ratio/low_mean": 0.06572701199911535, "clip_ratio/low_min": 0.06572701199911535, "clip_ratio/high_mean": 0.042327309027314186, "clip_ratio/high_max": 0.042327309027314186, "clip_ratio/region_mean": 0.10805432102642953, "reward_total_mean": 0.6166548728942871, "reward_meter_mean": 0.7288707494735718, "reward_meter_std": 0.32933905720710754, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.24800792336463928, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6166548728942871, "reward_total_composite_std": 0.34062808752059937, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 256.0} {"timestamp_utc": "2026-04-11T19:53:42Z", "mode": "train", "global_step": 257, "epoch": 0.009924312635156009, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 9.224242424242424e-06, "num_tokens": 557386.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.6820703744888306, "rewards/meter/std": 0.41173163056373596, "rewards/count_adherence/mean": 0.18269231915473938, "rewards/count_adherence/std": 0.05723259970545769, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.10419389605522156, "rewards/total_composite/std": 0.1085948497056961, "reward": 0.10419389605522156, "reward_std": 0.10859484225511551, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.10419389605522156, "reward_meter_mean": 0.6820703744888306, "reward_meter_std": 0.41173163056373596, "reward_count_adherence_mean": 0.18269231915473938, "reward_count_adherence_std": 0.05723259970545769, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.10419389605522156, "reward_total_composite_std": 0.1085948497056961, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 257.0} {"timestamp_utc": "2026-04-11T19:53:51Z", "mode": "train", "global_step": 258, "epoch": 0.009962928637627433, "loss": -0.0052, "grad_norm": 4.445526123046875, "learning_rate": 9.221212121212123e-06, "num_tokens": 559259.0, "completions/mean_length": 133.125, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 79.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.5733773708343506, "rewards/meter/std": 0.37895193696022034, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.5477786660194397, "rewards/total_composite/std": 0.4160669445991516, "reward": 0.5477786660194397, "reward_std": 0.4160669445991516, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.197053000330925, "sampling/sampling_logp_difference/max": 1.6546440124511719, "sampling/importance_sampling_ratio/min": 0.19116009771823883, "sampling/importance_sampling_ratio/mean": 1.0478163957595825, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3317447006702423, "clip_ratio/low_mean": 0.03242044895887375, "clip_ratio/low_min": 0.03242044895887375, "clip_ratio/high_mean": 0.09665837325155735, "clip_ratio/high_max": 0.09665837325155735, "clip_ratio/region_mean": 0.1290788222104311, "reward_total_mean": 0.5477786660194397, "reward_meter_mean": 0.5733773708343506, "reward_meter_std": 0.37895193696022034, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.5477786660194397, "reward_total_composite_std": 0.4160669445991516, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 258.0} {"timestamp_utc": "2026-04-11T19:53:56Z", "mode": "train", "global_step": 259, "epoch": 0.010001544640098857, "loss": 0.0631, "grad_norm": 11.931998252868652, "learning_rate": 9.21818181818182e-06, "num_tokens": 561151.0, "completions/mean_length": 64.5, "completions/min_length": 54.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.7427696585655212, "rewards/meter/std": 0.428190141916275, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7427696585655212, "rewards/total_composite/std": 0.428190141916275, "reward": 0.7427696585655212, "reward_std": 0.4281901717185974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19922052323818207, "sampling/sampling_logp_difference/max": 1.394083023071289, "sampling/importance_sampling_ratio/min": 0.248060405254364, "sampling/importance_sampling_ratio/mean": 1.0478657484054565, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7183807939291, "clip_ratio/low_mean": 0.031775956973433495, "clip_ratio/low_min": 0.031775956973433495, "clip_ratio/high_mean": 0.12396811135113239, "clip_ratio/high_max": 0.12396811135113239, "clip_ratio/region_mean": 0.1557440683245659, "reward_total_mean": 0.7427696585655212, "reward_meter_mean": 0.7427696585655212, "reward_meter_std": 0.428190141916275, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7427696585655212, "reward_total_composite_std": 0.428190141916275, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 259.0} {"timestamp_utc": "2026-04-11T19:54:01Z", "mode": "train", "global_step": 260, "epoch": 0.010040160642570281, "loss": 0.0645, "grad_norm": 8.251180648803711, "learning_rate": 9.215151515151515e-06, "num_tokens": 563063.0, "completions/mean_length": 78.0, "completions/min_length": 66.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.9256542325019836, "rewards/meter/std": 0.16513100266456604, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7421708106994629, "rewards/total_composite/std": 0.3694427013397217, "reward": 0.7421708106994629, "reward_std": 0.3694427013397217, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18370838463306427, "sampling/sampling_logp_difference/max": 1.8640003204345703, "sampling/importance_sampling_ratio/min": 0.15505114197731018, "sampling/importance_sampling_ratio/mean": 1.0489428043365479, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.69622403383255, "clip_ratio/low_mean": 0.04702932108193636, "clip_ratio/low_min": 0.04702932108193636, "clip_ratio/high_mean": 0.09239057265222073, "clip_ratio/high_max": 0.09239057265222073, "clip_ratio/region_mean": 0.13941989373415709, "reward_total_mean": 0.7421708106994629, "reward_meter_mean": 0.9256542325019836, "reward_meter_std": 0.16513100266456604, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7421708106994629, "reward_total_composite_std": 0.3694427013397217, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 260.0} {"timestamp_utc": "2026-04-11T19:54:05Z", "mode": "train", "global_step": 261, "epoch": 0.010078776645041705, "loss": -0.0436, "grad_norm": 23.678274154663086, "learning_rate": 9.212121212121213e-06, "num_tokens": 564601.0, "completions/mean_length": 34.25, "completions/min_length": 27.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9117841720581055, "rewards/meter/std": 0.18696558475494385, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9117841720581055, "rewards/total_composite/std": 0.18696558475494385, "reward": 0.9117841720581055, "reward_std": 0.18696556985378265, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20806393027305603, "sampling/sampling_logp_difference/max": 1.5957603454589844, "sampling/importance_sampling_ratio/min": 0.2197856903076172, "sampling/importance_sampling_ratio/mean": 1.0213043689727783, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.57423797249794, "clip_ratio/low_mean": 0.023148147389292717, "clip_ratio/low_min": 0.023148147389292717, "clip_ratio/high_mean": 0.18774798419326544, "clip_ratio/high_max": 0.18774798419326544, "clip_ratio/region_mean": 0.21089613158255816, "reward_total_mean": 0.9117841720581055, "reward_meter_mean": 0.9117841720581055, "reward_meter_std": 0.18696558475494385, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9117841720581055, "reward_total_composite_std": 0.18696558475494385, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 261.0} {"timestamp_utc": "2026-04-11T19:54:10Z", "mode": "train", "global_step": 262, "epoch": 0.01011739264751313, "loss": 0.027, "grad_norm": 8.469246864318848, "learning_rate": 9.20909090909091e-06, "num_tokens": 566352.0, "completions/mean_length": 67.875, "completions/min_length": 61.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.875, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.7471080422401428, "rewards/meter/std": 0.37732023000717163, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.746609091758728, "rewards/total_composite/std": 0.3784443438053131, "reward": 0.746609091758728, "reward_std": 0.37844428420066833, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.180844247341156, "sampling/sampling_logp_difference/max": 1.2776460647583008, "sampling/importance_sampling_ratio/min": 0.2786925435066223, "sampling/importance_sampling_ratio/mean": 1.0465917587280273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2145213186740875, "clip_ratio/low_mean": 0.04403977282345295, "clip_ratio/low_min": 0.04403977282345295, "clip_ratio/high_mean": 0.09604987595230341, "clip_ratio/high_max": 0.09604987595230341, "clip_ratio/region_mean": 0.14008964877575636, "reward_total_mean": 0.746609091758728, "reward_meter_mean": 0.7471080422401428, "reward_meter_std": 0.37732023000717163, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.746609091758728, "reward_total_composite_std": 0.3784443438053131, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 262.0} {"timestamp_utc": "2026-04-11T19:54:19Z", "mode": "train", "global_step": 263, "epoch": 0.010156008649984553, "loss": -0.0374, "grad_norm": 1.5207958221435547, "learning_rate": 9.206060606060607e-06, "num_tokens": 568030.0, "completions/mean_length": 364.75, "completions/min_length": 108.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 119.33333587646484, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.548949122428894, "rewards/meter/std": 0.3854943811893463, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.18898223340511322, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.4171389043331146, "rewards/total_composite/std": 0.4122898578643799, "reward": 0.4171389043331146, "reward_std": 0.4122898578643799, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14995871484279633, "sampling/sampling_logp_difference/max": 0.8275766372680664, "sampling/importance_sampling_ratio/min": 0.43710729479789734, "sampling/importance_sampling_ratio/mean": 1.0653656721115112, "sampling/importance_sampling_ratio/max": 1.990702748298645, "entropy": 0.768448993563652, "clip_ratio/low_mean": 0.007675438653677702, "clip_ratio/low_min": 0.007675438653677702, "clip_ratio/high_mean": 0.022365196608006954, "clip_ratio/high_max": 0.022365196608006954, "clip_ratio/region_mean": 0.030040635261684656, "reward_total_mean": 0.4171389043331146, "reward_meter_mean": 0.548949122428894, "reward_meter_std": 0.3854943811893463, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.18898223340511322, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.4171389043331146, "reward_total_composite_std": 0.4122898578643799, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 263.0} {"timestamp_utc": "2026-04-11T19:54:29Z", "mode": "train", "global_step": 264, "epoch": 0.010194624652455977, "loss": 0.0244, "grad_norm": 3.842879056930542, "learning_rate": 9.203030303030304e-06, "num_tokens": 570502.0, "completions/mean_length": 190.0, "completions/min_length": 101.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 144.0, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 172.0, "rewards/meter/mean": 0.3928939700126648, "rewards/meter/std": 0.41266870498657227, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.18898223340511322, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.2761991024017334, "rewards/total_composite/std": 0.37751540541648865, "reward": 0.2761991024017334, "reward_std": 0.37751540541648865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15981298685073853, "sampling/sampling_logp_difference/max": 1.371291160583496, "sampling/importance_sampling_ratio/min": 0.2537790536880493, "sampling/importance_sampling_ratio/mean": 1.032580018043518, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.793292224407196, "clip_ratio/low_mean": 0.04904576577246189, "clip_ratio/low_min": 0.04904576577246189, "clip_ratio/high_mean": 0.052887228317558765, "clip_ratio/high_max": 0.052887228317558765, "clip_ratio/region_mean": 0.10193299409002066, "reward_total_mean": 0.2761991024017334, "reward_meter_mean": 0.3928939700126648, "reward_meter_std": 0.41266870498657227, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.18898223340511322, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.2761991024017334, "reward_total_composite_std": 0.37751540541648865, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 264.0} {"timestamp_utc": "2026-04-11T19:54:38Z", "mode": "train", "global_step": 265, "epoch": 0.010233240654927402, "loss": -0.0242, "grad_norm": 3.180842876434326, "learning_rate": 9.200000000000002e-06, "num_tokens": 572168.0, "completions/mean_length": 184.25, "completions/min_length": 65.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 75.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.3811070919036865, "rewards/meter/std": 0.4830027222633362, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.3810850977897644, "rewards/total_composite/std": 0.483022540807724, "reward": 0.3810850977897644, "reward_std": 0.483022540807724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17107190191745758, "sampling/sampling_logp_difference/max": 1.190293312072754, "sampling/importance_sampling_ratio/min": 0.30413204431533813, "sampling/importance_sampling_ratio/mean": 1.0414031744003296, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5678513646125793, "clip_ratio/low_mean": 0.047125319950282574, "clip_ratio/low_min": 0.047125319950282574, "clip_ratio/high_mean": 0.05313897877931595, "clip_ratio/high_max": 0.05313897877931595, "clip_ratio/region_mean": 0.10026429872959852, "reward_total_mean": 0.3810850977897644, "reward_meter_mean": 0.3811070919036865, "reward_meter_std": 0.4830027222633362, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.3810850977897644, "reward_total_composite_std": 0.483022540807724, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 265.0} {"timestamp_utc": "2026-04-11T19:54:48Z", "mode": "train", "global_step": 266, "epoch": 0.010271856657398826, "loss": -0.2009, "grad_norm": 2.051448345184326, "learning_rate": 9.196969696969697e-06, "num_tokens": 574908.0, "completions/mean_length": 452.5, "completions/min_length": 309.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 353.3333435058594, "completions/min_terminated_length": 309.0, "completions/max_terminated_length": 397.0, "rewards/meter/mean": 0.2717985510826111, "rewards/meter/std": 0.30201706290245056, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.1836577206850052, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/total_composite/mean": 0.14004071056842804, "rewards/total_composite/std": 0.255687415599823, "reward": 0.14004071056842804, "reward_std": 0.2556873857975006, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15083111822605133, "sampling/sampling_logp_difference/max": 1.1689472198486328, "sampling/importance_sampling_ratio/min": 0.3106938600540161, "sampling/importance_sampling_ratio/mean": 1.0511150360107422, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6620796918869019, "clip_ratio/low_mean": 0.01574307307600975, "clip_ratio/low_min": 0.01574307307600975, "clip_ratio/high_mean": 0.021464127115905285, "clip_ratio/high_max": 0.021464127115905285, "clip_ratio/region_mean": 0.037207200191915035, "reward_total_mean": 0.14004071056842804, "reward_meter_mean": 0.2717985510826111, "reward_meter_std": 0.30201706290245056, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.1836577206850052, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_total_composite_mean": 0.14004071056842804, "reward_total_composite_std": 0.255687415599823, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 266.0} {"timestamp_utc": "2026-04-11T19:54:53Z", "mode": "train", "global_step": 267, "epoch": 0.01031047265987025, "loss": 0.1645, "grad_norm": 10.749847412109375, "learning_rate": 9.193939393939395e-06, "num_tokens": 576616.0, "completions/mean_length": 70.5, "completions/min_length": 53.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.7857816219329834, "rewards/meter/std": 0.35958775877952576, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7857815027236938, "rewards/total_composite/std": 0.359588086605072, "reward": 0.7857815027236938, "reward_std": 0.35958805680274963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19864748418331146, "sampling/sampling_logp_difference/max": 1.2832155227661133, "sampling/importance_sampling_ratio/min": 0.2771447002887726, "sampling/importance_sampling_ratio/mean": 1.053087830543518, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6592743396759033, "clip_ratio/low_mean": 0.04466869868338108, "clip_ratio/low_min": 0.04466869868338108, "clip_ratio/high_mean": 0.12783648911863565, "clip_ratio/high_max": 0.12783648911863565, "clip_ratio/region_mean": 0.17250518780201674, "reward_total_mean": 0.7857815027236938, "reward_meter_mean": 0.7857816219329834, "reward_meter_std": 0.35958775877952576, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7857815027236938, "reward_total_composite_std": 0.359588086605072, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 267.0} {"timestamp_utc": "2026-04-11T19:54:58Z", "mode": "train", "global_step": 268, "epoch": 0.010349088662341674, "loss": 0.0142, "grad_norm": 18.469411849975586, "learning_rate": 9.190909090909092e-06, "num_tokens": 578371.0, "completions/mean_length": 51.375, "completions/min_length": 43.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.6911875009536743, "rewards/meter/std": 0.36712267994880676, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6911875009536743, "rewards/total_composite/std": 0.36712267994880676, "reward": 0.6911875009536743, "reward_std": 0.3671226501464844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23253513872623444, "sampling/sampling_logp_difference/max": 3.009150981903076, "sampling/importance_sampling_ratio/min": 0.04933354631066322, "sampling/importance_sampling_ratio/mean": 0.980422854423523, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0063387379050255, "clip_ratio/low_mean": 0.06932727806270123, "clip_ratio/low_min": 0.06932727806270123, "clip_ratio/high_mean": 0.1120001059025526, "clip_ratio/high_max": 0.1120001059025526, "clip_ratio/region_mean": 0.18132738396525383, "reward_total_mean": 0.6911875009536743, "reward_meter_mean": 0.6911875009536743, "reward_meter_std": 0.36712267994880676, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6911875009536743, "reward_total_composite_std": 0.36712267994880676, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 268.0} {"timestamp_utc": "2026-04-11T19:55:03Z", "mode": "train", "global_step": 269, "epoch": 0.010387704664813098, "loss": 0.0128, "grad_norm": 7.486761569976807, "learning_rate": 9.187878787878789e-06, "num_tokens": 580441.0, "completions/mean_length": 92.75, "completions/min_length": 83.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.75, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.8626927137374878, "rewards/meter/std": 0.2941436171531677, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7411905527114868, "rewards/total_composite/std": 0.41744595766067505, "reward": 0.7411905527114868, "reward_std": 0.41744592785835266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1645209938287735, "sampling/sampling_logp_difference/max": 1.7359848022460938, "sampling/importance_sampling_ratio/min": 0.17622657120227814, "sampling/importance_sampling_ratio/mean": 1.0476194620132446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.058430328965187, "clip_ratio/low_mean": 0.049313412979245186, "clip_ratio/low_min": 0.049313412979245186, "clip_ratio/high_mean": 0.09056044183671474, "clip_ratio/high_max": 0.09056044183671474, "clip_ratio/region_mean": 0.13987385481595993, "reward_total_mean": 0.7411905527114868, "reward_meter_mean": 0.8626927137374878, "reward_meter_std": 0.2941436171531677, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7411905527114868, "reward_total_composite_std": 0.41744595766067505, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 269.0} {"timestamp_utc": "2026-04-11T19:55:13Z", "mode": "train", "global_step": 270, "epoch": 0.010426320667284522, "loss": -0.0971, "grad_norm": 1.4978303909301758, "learning_rate": 9.184848484848485e-06, "num_tokens": 582411.0, "completions/mean_length": 445.25, "completions/min_length": 228.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.75, "completions/mean_terminated_length": 245.0, "completions/min_terminated_length": 228.0, "completions/max_terminated_length": 262.0, "rewards/meter/mean": 0.07851605117321014, "rewards/meter/std": 0.06109999120235443, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.030055876821279526, "rewards/total_composite/std": 0.040369123220443726, "reward": 0.030055876821279526, "reward_std": 0.040369123220443726, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1704261153936386, "sampling/sampling_logp_difference/max": 1.4844207763671875, "sampling/importance_sampling_ratio/min": 0.22663357853889465, "sampling/importance_sampling_ratio/mean": 1.0170917510986328, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48567473888397217, "clip_ratio/low_mean": 0.013358778320252895, "clip_ratio/low_min": 0.013358778320252895, "clip_ratio/high_mean": 0.019736841320991516, "clip_ratio/high_max": 0.019736841320991516, "clip_ratio/region_mean": 0.03309561964124441, "reward_total_mean": 0.030055876821279526, "reward_meter_mean": 0.07851605117321014, "reward_meter_std": 0.06109999120235443, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.030055876821279526, "reward_total_composite_std": 0.040369123220443726, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 270.0} {"timestamp_utc": "2026-04-11T19:55:17Z", "mode": "train", "global_step": 271, "epoch": 0.010464936669755946, "loss": 0.0604, "grad_norm": 10.39463996887207, "learning_rate": 9.181818181818184e-06, "num_tokens": 584259.0, "completions/mean_length": 59.0, "completions/min_length": 42.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.6009393930435181, "rewards/meter/std": 0.41454699635505676, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6009393930435181, "rewards/total_composite/std": 0.41454699635505676, "reward": 0.6009393930435181, "reward_std": 0.41454702615737915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16193664073944092, "sampling/sampling_logp_difference/max": 0.9821362495422363, "sampling/importance_sampling_ratio/min": 0.3745101988315582, "sampling/importance_sampling_ratio/mean": 1.017964482307434, "sampling/importance_sampling_ratio/max": 1.8810900449752808, "entropy": 1.7888787239789963, "clip_ratio/low_mean": 0.07794231548905373, "clip_ratio/low_min": 0.07794231548905373, "clip_ratio/high_mean": 0.09404239989817142, "clip_ratio/high_max": 0.09404239989817142, "clip_ratio/region_mean": 0.17198471538722515, "reward_total_mean": 0.6009393930435181, "reward_meter_mean": 0.6009393930435181, "reward_meter_std": 0.41454699635505676, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6009393930435181, "reward_total_composite_std": 0.41454699635505676, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 271.0} {"timestamp_utc": "2026-04-11T19:55:22Z", "mode": "train", "global_step": 272, "epoch": 0.01050355267222737, "loss": 0.239, "grad_norm": 16.354007720947266, "learning_rate": 9.178787878787879e-06, "num_tokens": 585848.0, "completions/mean_length": 37.625, "completions/min_length": 29.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6360828876495361, "rewards/meter/std": 0.43516474962234497, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5806645154953003, "rewards/total_composite/std": 0.4882129728794098, "reward": 0.5806645154953003, "reward_std": 0.4882129728794098, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1913789063692093, "sampling/sampling_logp_difference/max": 1.653336524963379, "sampling/importance_sampling_ratio/min": 0.19141019880771637, "sampling/importance_sampling_ratio/mean": 1.0565941333770752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1982433944940567, "clip_ratio/low_mean": 0.059118627570569515, "clip_ratio/low_min": 0.059118627570569515, "clip_ratio/high_mean": 0.07653909968212247, "clip_ratio/high_max": 0.07653909968212247, "clip_ratio/region_mean": 0.13565772725269198, "reward_total_mean": 0.5806645154953003, "reward_meter_mean": 0.6360828876495361, "reward_meter_std": 0.43516474962234497, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5806645154953003, "reward_total_composite_std": 0.4882129728794098, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 272.0} {"timestamp_utc": "2026-04-11T19:55:32Z", "mode": "train", "global_step": 273, "epoch": 0.010542168674698794, "loss": -0.2266, "grad_norm": 1.6592974662780762, "learning_rate": 9.175757575757576e-06, "num_tokens": 588010.0, "completions/mean_length": 510.25, "completions/min_length": 498.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.875, "completions/mean_terminated_length": 498.0, "completions/min_terminated_length": 498.0, "completions/max_terminated_length": 498.0, "rewards/meter/mean": 0.31048354506492615, "rewards/meter/std": 0.3288111686706543, "rewards/count_adherence/mean": 0.4821428656578064, "rewards/count_adherence/std": 0.23458294570446014, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/total_composite/mean": 0.11888930201530457, "rewards/total_composite/std": 0.20482075214385986, "reward": 0.11888930201530457, "reward_std": 0.20482073724269867, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08662550151348114, "sampling/sampling_logp_difference/max": 1.0271673202514648, "sampling/importance_sampling_ratio/min": 0.3580196797847748, "sampling/importance_sampling_ratio/mean": 1.0267904996871948, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12209701538085938, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006777108646929264, "clip_ratio/high_max": 0.006777108646929264, "clip_ratio/region_mean": 0.006777108646929264, "reward_total_mean": 0.11888930201530457, "reward_meter_mean": 0.31048354506492615, "reward_meter_std": 0.3288111686706543, "reward_count_adherence_mean": 0.4821428656578064, "reward_count_adherence_std": 0.23458294570446014, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_total_composite_mean": 0.11888930201530457, "reward_total_composite_std": 0.20482075214385986, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 273.0} {"timestamp_utc": "2026-04-11T19:55:42Z", "mode": "train", "global_step": 274, "epoch": 0.010580784677170219, "loss": -0.1042, "grad_norm": 3.6858744621276855, "learning_rate": 9.172727272727274e-06, "num_tokens": 590872.0, "completions/mean_length": 229.75, "completions/min_length": 160.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 189.42857360839844, "completions/min_terminated_length": 160.0, "completions/max_terminated_length": 213.0, "rewards/meter/mean": 0.46854132413864136, "rewards/meter/std": 0.38436928391456604, "rewards/count_adherence/mean": 0.8035714626312256, "rewards/count_adherence/std": 0.15152288973331451, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.39238011837005615, "rewards/total_composite/std": 0.34078967571258545, "reward": 0.39238011837005615, "reward_std": 0.34078967571258545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14069926738739014, "sampling/sampling_logp_difference/max": 1.1839194297790527, "sampling/importance_sampling_ratio/min": 0.3060767352581024, "sampling/importance_sampling_ratio/mean": 1.0281857252120972, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.431438848376274, "clip_ratio/low_mean": 0.024902896489948034, "clip_ratio/low_min": 0.024902896489948034, "clip_ratio/high_mean": 0.06644443050026894, "clip_ratio/high_max": 0.06644443050026894, "clip_ratio/region_mean": 0.09134732699021697, "reward_total_mean": 0.39238011837005615, "reward_meter_mean": 0.46854132413864136, "reward_meter_std": 0.38436928391456604, "reward_count_adherence_mean": 0.8035714626312256, "reward_count_adherence_std": 0.15152288973331451, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.39238011837005615, "reward_total_composite_std": 0.34078967571258545, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 274.0} {"timestamp_utc": "2026-04-11T19:55:51Z", "mode": "train", "global_step": 275, "epoch": 0.010619400679641643, "loss": -0.1344, "grad_norm": 3.059723138809204, "learning_rate": 9.169696969696971e-06, "num_tokens": 592749.0, "completions/mean_length": 198.625, "completions/min_length": 81.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 94.16667175292969, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.6977724432945251, "rewards/meter/std": 0.38409724831581116, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5918634533882141, "rewards/total_composite/std": 0.44841268658638, "reward": 0.5918634533882141, "reward_std": 0.4484126567840576, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17513571679592133, "sampling/sampling_logp_difference/max": 1.2321577072143555, "sampling/importance_sampling_ratio/min": 0.2916625738143921, "sampling/importance_sampling_ratio/mean": 1.050614833831787, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5135751068592072, "clip_ratio/low_mean": 0.017326733097434044, "clip_ratio/low_min": 0.017326733097434044, "clip_ratio/high_mean": 0.0879957852885127, "clip_ratio/high_max": 0.0879957852885127, "clip_ratio/region_mean": 0.10532251838594675, "reward_total_mean": 0.5918634533882141, "reward_meter_mean": 0.6977724432945251, "reward_meter_std": 0.38409724831581116, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5918634533882141, "reward_total_composite_std": 0.44841268658638, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 275.0} {"timestamp_utc": "2026-04-11T19:55:57Z", "mode": "train", "global_step": 276, "epoch": 0.010658016682113068, "loss": 0.062, "grad_norm": 6.9882683753967285, "learning_rate": 9.166666666666666e-06, "num_tokens": 594942.0, "completions/mean_length": 116.125, "completions/min_length": 102.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.125, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.8023958206176758, "rewards/meter/std": 0.32191282510757446, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.47085532546043396, "rewards/total_composite/std": 0.4658812880516052, "reward": 0.47085532546043396, "reward_std": 0.46588125824928284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18583007156848907, "sampling/sampling_logp_difference/max": 1.5221500396728516, "sampling/importance_sampling_ratio/min": 0.21824216842651367, "sampling/importance_sampling_ratio/mean": 1.0579421520233154, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.8212269842624664, "clip_ratio/low_mean": 0.05223934445530176, "clip_ratio/low_min": 0.05223934445530176, "clip_ratio/high_mean": 0.06428467482328415, "clip_ratio/high_max": 0.06428467482328415, "clip_ratio/region_mean": 0.11652401927858591, "reward_total_mean": 0.47085532546043396, "reward_meter_mean": 0.8023958206176758, "reward_meter_std": 0.32191282510757446, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.47085532546043396, "reward_total_composite_std": 0.4658812880516052, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 276.0} {"timestamp_utc": "2026-04-11T19:56:01Z", "mode": "train", "global_step": 277, "epoch": 0.010696632684584493, "loss": 0.0701, "grad_norm": 16.393516540527344, "learning_rate": 9.163636363636365e-06, "num_tokens": 596617.0, "completions/mean_length": 46.375, "completions/min_length": 38.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.375, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.5784964561462402, "rewards/meter/std": 0.3392086327075958, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5784964561462402, "rewards/total_composite/std": 0.3392086327075958, "reward": 0.5784964561462402, "reward_std": 0.3392086327075958, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2193935364484787, "sampling/sampling_logp_difference/max": 2.0590333938598633, "sampling/importance_sampling_ratio/min": 0.12757723033428192, "sampling/importance_sampling_ratio/mean": 1.0258041620254517, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2117373794317245, "clip_ratio/low_mean": 0.04722222313284874, "clip_ratio/low_min": 0.04722222313284874, "clip_ratio/high_mean": 0.12392107397317886, "clip_ratio/high_max": 0.12392107397317886, "clip_ratio/region_mean": 0.1711432971060276, "reward_total_mean": 0.5784964561462402, "reward_meter_mean": 0.5784964561462402, "reward_meter_std": 0.3392086327075958, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5784964561462402, "reward_total_composite_std": 0.3392086327075958, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 277.0} {"timestamp_utc": "2026-04-11T19:56:06Z", "mode": "train", "global_step": 278, "epoch": 0.010735248687055917, "loss": 0.0379, "grad_norm": 7.285430431365967, "learning_rate": 9.160606060606061e-06, "num_tokens": 598870.0, "completions/mean_length": 111.625, "completions/min_length": 99.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.625, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.8526768684387207, "rewards/meter/std": 0.27573806047439575, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8526768684387207, "rewards/total_composite/std": 0.27573806047439575, "reward": 0.8526768684387207, "reward_std": 0.27573806047439575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14818187057971954, "sampling/sampling_logp_difference/max": 1.2836666107177734, "sampling/importance_sampling_ratio/min": 0.2770197093486786, "sampling/importance_sampling_ratio/mean": 1.0367151498794556, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7266891598701477, "clip_ratio/low_mean": 0.03383508883416653, "clip_ratio/low_min": 0.03383508883416653, "clip_ratio/high_mean": 0.0827424954622984, "clip_ratio/high_max": 0.0827424954622984, "clip_ratio/region_mean": 0.11657758429646492, "reward_total_mean": 0.8526768684387207, "reward_meter_mean": 0.8526768684387207, "reward_meter_std": 0.27573806047439575, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8526768684387207, "reward_total_composite_std": 0.27573806047439575, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 278.0} {"timestamp_utc": "2026-04-11T19:56:11Z", "mode": "train", "global_step": 279, "epoch": 0.01077386468952734, "loss": 0.0423, "grad_norm": 11.760062217712402, "learning_rate": 9.157575757575758e-06, "num_tokens": 600549.0, "completions/mean_length": 59.875, "completions/min_length": 56.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.45295605063438416, "rewards/meter/std": 0.416568398475647, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.45295605063438416, "rewards/total_composite/std": 0.416568398475647, "reward": 0.45295605063438416, "reward_std": 0.416568398475647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16330303251743317, "sampling/sampling_logp_difference/max": 1.0449113845825195, "sampling/importance_sampling_ratio/min": 0.35172298550605774, "sampling/importance_sampling_ratio/mean": 1.029968023300171, "sampling/importance_sampling_ratio/max": 1.9151904582977295, "entropy": 1.8726500868797302, "clip_ratio/low_mean": 0.08847619779407978, "clip_ratio/low_min": 0.08847619779407978, "clip_ratio/high_mean": 0.05927513726055622, "clip_ratio/high_max": 0.05927513726055622, "clip_ratio/region_mean": 0.147751335054636, "reward_total_mean": 0.45295605063438416, "reward_meter_mean": 0.45295605063438416, "reward_meter_std": 0.416568398475647, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.45295605063438416, "reward_total_composite_std": 0.416568398475647, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 279.0} {"timestamp_utc": "2026-04-11T19:56:17Z", "mode": "train", "global_step": 280, "epoch": 0.010812480691998765, "loss": 0.0106, "grad_norm": 7.997870922088623, "learning_rate": 9.154545454545455e-06, "num_tokens": 603365.0, "completions/mean_length": 154.0, "completions/min_length": 125.0, "completions/max_length": 187.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.0, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 187.0, "rewards/meter/mean": 0.4514502286911011, "rewards/meter/std": 0.3197373151779175, "rewards/count_adherence/mean": 0.8958333134651184, "rewards/count_adherence/std": 0.08625820279121399, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.404763400554657, "rewards/total_composite/std": 0.32716503739356995, "reward": 0.404763400554657, "reward_std": 0.32716503739356995, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1659134477376938, "sampling/sampling_logp_difference/max": 1.630927562713623, "sampling/importance_sampling_ratio/min": 0.19574791193008423, "sampling/importance_sampling_ratio/mean": 1.0190558433532715, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5907130688428879, "clip_ratio/low_mean": 0.06699808966368437, "clip_ratio/low_min": 0.06699808966368437, "clip_ratio/high_mean": 0.08833360858261585, "clip_ratio/high_max": 0.08833360858261585, "clip_ratio/region_mean": 0.15533169824630022, "reward_total_mean": 0.404763400554657, "reward_meter_mean": 0.4514502286911011, "reward_meter_std": 0.3197373151779175, "reward_count_adherence_mean": 0.8958333134651184, "reward_count_adherence_std": 0.08625820279121399, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.404763400554657, "reward_total_composite_std": 0.32716503739356995, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 280.0} {"timestamp_utc": "2026-04-11T19:56:26Z", "mode": "train", "global_step": 281, "epoch": 0.010851096694470189, "loss": -0.1226, "grad_norm": 1.4875062704086304, "learning_rate": 9.151515151515153e-06, "num_tokens": 605333.0, "completions/mean_length": 205.0, "completions/min_length": 79.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 102.66667175292969, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.8328183889389038, "rewards/meter/std": 0.3320009410381317, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8265775442123413, "rewards/total_composite/std": 0.3488611578941345, "reward": 0.8265775442123413, "reward_std": 0.34886112809181213, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17393165826797485, "sampling/sampling_logp_difference/max": 1.8032541275024414, "sampling/importance_sampling_ratio/min": 0.16476187109947205, "sampling/importance_sampling_ratio/mean": 1.0586100816726685, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8442810326814651, "clip_ratio/low_mean": 0.02150537632405758, "clip_ratio/low_min": 0.02150537632405758, "clip_ratio/high_mean": 0.07967114355415106, "clip_ratio/high_max": 0.07967114355415106, "clip_ratio/region_mean": 0.10117651987820864, "reward_total_mean": 0.8265775442123413, "reward_meter_mean": 0.8328183889389038, "reward_meter_std": 0.3320009410381317, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8265775442123413, "reward_total_composite_std": 0.3488611578941345, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 281.0} {"timestamp_utc": "2026-04-11T19:56:36Z", "mode": "train", "global_step": 282, "epoch": 0.010889712696941613, "loss": -0.2053, "grad_norm": 2.2956902980804443, "learning_rate": 9.148484848484848e-06, "num_tokens": 607958.0, "completions/mean_length": 308.125, "completions/min_length": 179.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 185.8000030517578, "completions/min_terminated_length": 179.0, "completions/max_terminated_length": 193.0, "rewards/meter/mean": 0.3302040994167328, "rewards/meter/std": 0.34381669759750366, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.15430334210395813, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.27106142044067383, "rewards/total_composite/std": 0.29077771306037903, "reward": 0.27106142044067383, "reward_std": 0.29077771306037903, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1307448148727417, "sampling/sampling_logp_difference/max": 1.9015026092529297, "sampling/importance_sampling_ratio/min": 0.1493440419435501, "sampling/importance_sampling_ratio/mean": 1.0214406251907349, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0763673335313797, "clip_ratio/low_mean": 0.023917971178889275, "clip_ratio/low_min": 0.023917971178889275, "clip_ratio/high_mean": 0.05353208538144827, "clip_ratio/high_max": 0.05353208538144827, "clip_ratio/region_mean": 0.07745005656033754, "reward_total_mean": 0.27106142044067383, "reward_meter_mean": 0.3302040994167328, "reward_meter_std": 0.34381669759750366, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.15430334210395813, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.27106142044067383, "reward_total_composite_std": 0.29077771306037903, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 282.0} {"timestamp_utc": "2026-04-11T19:56:41Z", "mode": "train", "global_step": 283, "epoch": 0.010928328699413037, "loss": 0.0578, "grad_norm": 6.667464256286621, "learning_rate": 9.145454545454546e-06, "num_tokens": 610721.0, "completions/mean_length": 140.375, "completions/min_length": 120.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.375, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.8891458511352539, "rewards/meter/std": 0.24647405743598938, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7411341667175293, "rewards/total_composite/std": 0.1899271160364151, "reward": 0.7411341667175293, "reward_std": 0.1899271160364151, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15236106514930725, "sampling/sampling_logp_difference/max": 1.6330780982971191, "sampling/importance_sampling_ratio/min": 0.2201007455587387, "sampling/importance_sampling_ratio/mean": 1.0217300653457642, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7298996597528458, "clip_ratio/low_mean": 0.01697530783712864, "clip_ratio/low_min": 0.01697530783712864, "clip_ratio/high_mean": 0.11775326356291771, "clip_ratio/high_max": 0.11775326356291771, "clip_ratio/region_mean": 0.13472857140004635, "reward_total_mean": 0.7411341667175293, "reward_meter_mean": 0.8891458511352539, "reward_meter_std": 0.24647405743598938, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7411341667175293, "reward_total_composite_std": 0.1899271160364151, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 283.0} {"timestamp_utc": "2026-04-11T19:56:46Z", "mode": "train", "global_step": 284, "epoch": 0.010966944701884461, "loss": 0.0181, "grad_norm": 7.729722499847412, "learning_rate": 9.142424242424243e-06, "num_tokens": 612706.0, "completions/mean_length": 96.125, "completions/min_length": 88.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.125, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9732859134674072, "rewards/meter/std": 0.022573497146368027, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9732859134674072, "rewards/total_composite/std": 0.022573497146368027, "reward": 0.9732859134674072, "reward_std": 0.02257349155843258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14632542431354523, "sampling/sampling_logp_difference/max": 1.1845827102661133, "sampling/importance_sampling_ratio/min": 0.3058738112449646, "sampling/importance_sampling_ratio/mean": 1.028389573097229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9220721423625946, "clip_ratio/low_mean": 0.03779107145965099, "clip_ratio/low_min": 0.03779107145965099, "clip_ratio/high_mean": 0.09173304960131645, "clip_ratio/high_max": 0.09173304960131645, "clip_ratio/region_mean": 0.12952412106096745, "reward_total_mean": 0.9732859134674072, "reward_meter_mean": 0.9732859134674072, "reward_meter_std": 0.022573497146368027, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9732859134674072, "reward_total_composite_std": 0.022573497146368027, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 284.0} {"timestamp_utc": "2026-04-11T19:56:51Z", "mode": "train", "global_step": 285, "epoch": 0.011005560704355885, "loss": 0.0231, "grad_norm": 10.306532859802246, "learning_rate": 9.13939393939394e-06, "num_tokens": 614381.0, "completions/mean_length": 59.375, "completions/min_length": 52.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.375, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.47206586599349976, "rewards/meter/std": 0.39846837520599365, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.47206586599349976, "rewards/total_composite/std": 0.39846837520599365, "reward": 0.47206586599349976, "reward_std": 0.39846834540367126, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16083520650863647, "sampling/sampling_logp_difference/max": 1.7305299043655396, "sampling/importance_sampling_ratio/min": 0.17719048261642456, "sampling/importance_sampling_ratio/mean": 1.0313324928283691, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.934165135025978, "clip_ratio/low_mean": 0.0852982671931386, "clip_ratio/low_min": 0.0852982671931386, "clip_ratio/high_mean": 0.05674981512129307, "clip_ratio/high_max": 0.05674981512129307, "clip_ratio/region_mean": 0.14204808231443167, "reward_total_mean": 0.47206586599349976, "reward_meter_mean": 0.47206586599349976, "reward_meter_std": 0.39846837520599365, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.47206586599349976, "reward_total_composite_std": 0.39846837520599365, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 285.0} {"timestamp_utc": "2026-04-11T19:56:57Z", "mode": "train", "global_step": 286, "epoch": 0.01104417670682731, "loss": 0.1343, "grad_norm": 8.738157272338867, "learning_rate": 9.136363636363637e-06, "num_tokens": 617527.0, "completions/mean_length": 177.25, "completions/min_length": 149.0, "completions/max_length": 225.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 177.25, "completions/min_terminated_length": 149.0, "completions/max_terminated_length": 225.0, "rewards/meter/mean": 0.4146118462085724, "rewards/meter/std": 0.2753751873970032, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.05050762742757797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3599608540534973, "rewards/total_composite/std": 0.2333701252937317, "reward": 0.3599608540534973, "reward_std": 0.23337014019489288, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17458735406398773, "sampling/sampling_logp_difference/max": 3.983931064605713, "sampling/importance_sampling_ratio/min": 0.0186123289167881, "sampling/importance_sampling_ratio/mean": 0.9998421669006348, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8629858121275902, "clip_ratio/low_mean": 0.0790464598685503, "clip_ratio/low_min": 0.0790464598685503, "clip_ratio/high_mean": 0.07367282547056675, "clip_ratio/high_max": 0.07367282547056675, "clip_ratio/region_mean": 0.15271928533911705, "reward_total_mean": 0.3599608540534973, "reward_meter_mean": 0.4146118462085724, "reward_meter_std": 0.2753751873970032, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.05050762742757797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3599608540534973, "reward_total_composite_std": 0.2333701252937317, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 286.0} {"timestamp_utc": "2026-04-11T19:57:03Z", "mode": "train", "global_step": 287, "epoch": 0.011082792709298734, "loss": -0.003, "grad_norm": 7.0826334953308105, "learning_rate": 9.133333333333335e-06, "num_tokens": 619875.0, "completions/mean_length": 129.5, "completions/min_length": 120.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.5, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.38049113750457764, "rewards/meter/std": 0.30704429745674133, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.30439293384552, "rewards/total_composite/std": 0.24563542008399963, "reward": 0.30439293384552, "reward_std": 0.24563542008399963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1677759736776352, "sampling/sampling_logp_difference/max": 1.4661445617675781, "sampling/importance_sampling_ratio/min": 0.230813667178154, "sampling/importance_sampling_ratio/mean": 1.029800295829773, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8472711890935898, "clip_ratio/low_mean": 0.1001923680305481, "clip_ratio/low_min": 0.1001923680305481, "clip_ratio/high_mean": 0.03732517547905445, "clip_ratio/high_max": 0.03732517547905445, "clip_ratio/region_mean": 0.13751754350960255, "reward_total_mean": 0.30439293384552, "reward_meter_mean": 0.38049113750457764, "reward_meter_std": 0.30704429745674133, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.30439293384552, "reward_total_composite_std": 0.24563542008399963, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 287.0} {"timestamp_utc": "2026-04-11T19:57:07Z", "mode": "train", "global_step": 288, "epoch": 0.011121408711770158, "loss": -0.0746, "grad_norm": 9.476400375366211, "learning_rate": 9.130303030303032e-06, "num_tokens": 621483.0, "completions/mean_length": 61.0, "completions/min_length": 50.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.6926716566085815, "rewards/meter/std": 0.32063576579093933, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6926716566085815, "rewards/total_composite/std": 0.32063576579093933, "reward": 0.6926716566085815, "reward_std": 0.32063573598861694, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1627330631017685, "sampling/sampling_logp_difference/max": 1.1543960571289062, "sampling/importance_sampling_ratio/min": 0.31524786353111267, "sampling/importance_sampling_ratio/mean": 1.02389657497406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9991831853985786, "clip_ratio/low_mean": 0.0725222835317254, "clip_ratio/low_min": 0.0725222835317254, "clip_ratio/high_mean": 0.07457270473241806, "clip_ratio/high_max": 0.07457270473241806, "clip_ratio/region_mean": 0.14709498826414347, "reward_total_mean": 0.6926716566085815, "reward_meter_mean": 0.6926716566085815, "reward_meter_std": 0.32063576579093933, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6926716566085815, "reward_total_composite_std": 0.32063576579093933, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 288.0} {"timestamp_utc": "2026-04-11T19:57:12Z", "mode": "train", "global_step": 289, "epoch": 0.011160024714241582, "loss": 0.093, "grad_norm": 10.661824226379395, "learning_rate": 9.127272727272727e-06, "num_tokens": 623327.0, "completions/mean_length": 56.5, "completions/min_length": 40.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.6810826063156128, "rewards/meter/std": 0.39801594614982605, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6810826063156128, "rewards/total_composite/std": 0.39801594614982605, "reward": 0.6810826063156128, "reward_std": 0.39801594614982605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12699994444847107, "sampling/sampling_logp_difference/max": 1.2867097854614258, "sampling/importance_sampling_ratio/min": 0.2761779725551605, "sampling/importance_sampling_ratio/mean": 1.0323865413665771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.427998036146164, "clip_ratio/low_mean": 0.03081388957798481, "clip_ratio/low_min": 0.03081388957798481, "clip_ratio/high_mean": 0.08544671628624201, "clip_ratio/high_max": 0.08544671628624201, "clip_ratio/region_mean": 0.11626060586422682, "reward_total_mean": 0.6810826063156128, "reward_meter_mean": 0.6810826063156128, "reward_meter_std": 0.39801594614982605, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6810826063156128, "reward_total_composite_std": 0.39801594614982605, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 289.0} {"timestamp_utc": "2026-04-11T19:57:17Z", "mode": "train", "global_step": 290, "epoch": 0.011198640716713006, "loss": -0.0217, "grad_norm": 8.565241813659668, "learning_rate": 9.124242424242425e-06, "num_tokens": 625439.0, "completions/mean_length": 91.0, "completions/min_length": 76.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.7161059379577637, "rewards/meter/std": 0.37380164861679077, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7161059379577637, "rewards/total_composite/std": 0.37380164861679077, "reward": 0.7161059379577637, "reward_std": 0.37380164861679077, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1450376659631729, "sampling/sampling_logp_difference/max": 1.4267511367797852, "sampling/importance_sampling_ratio/min": 0.24008767306804657, "sampling/importance_sampling_ratio/mean": 1.0219635963439941, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5821717530488968, "clip_ratio/low_mean": 0.036112773232162, "clip_ratio/low_min": 0.036112773232162, "clip_ratio/high_mean": 0.08129793964326382, "clip_ratio/high_max": 0.08129793964326382, "clip_ratio/region_mean": 0.11741071287542582, "reward_total_mean": 0.7161059379577637, "reward_meter_mean": 0.7161059379577637, "reward_meter_std": 0.37380164861679077, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7161059379577637, "reward_total_composite_std": 0.37380164861679077, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 290.0} {"timestamp_utc": "2026-04-11T19:57:22Z", "mode": "train", "global_step": 291, "epoch": 0.01123725671918443, "loss": 0.0824, "grad_norm": 9.695059776306152, "learning_rate": 9.121212121212122e-06, "num_tokens": 627518.0, "completions/mean_length": 84.875, "completions/min_length": 57.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.875, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.8615293502807617, "rewards/meter/std": 0.3134516179561615, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8615293502807617, "rewards/total_composite/std": 0.3134516179561615, "reward": 0.8615293502807617, "reward_std": 0.3134515881538391, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17841492593288422, "sampling/sampling_logp_difference/max": 1.3077259063720703, "sampling/importance_sampling_ratio/min": 0.27043434977531433, "sampling/importance_sampling_ratio/mean": 1.020525574684143, "sampling/importance_sampling_ratio/max": 1.9597175121307373, "entropy": 2.2144875079393387, "clip_ratio/low_mean": 0.019999999552965164, "clip_ratio/low_min": 0.019999999552965164, "clip_ratio/high_mean": 0.15774553548544645, "clip_ratio/high_max": 0.15774553548544645, "clip_ratio/region_mean": 0.17774553503841162, "reward_total_mean": 0.8615293502807617, "reward_meter_mean": 0.8615293502807617, "reward_meter_std": 0.3134516179561615, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8615293502807617, "reward_total_composite_std": 0.3134516179561615, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 291.0} {"timestamp_utc": "2026-04-11T19:57:28Z", "mode": "train", "global_step": 292, "epoch": 0.011275872721655854, "loss": 0.0485, "grad_norm": 4.9265546798706055, "learning_rate": 9.118181818181819e-06, "num_tokens": 630696.0, "completions/mean_length": 199.25, "completions/min_length": 179.0, "completions/max_length": 222.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 199.25, "completions/min_terminated_length": 179.0, "completions/max_terminated_length": 222.0, "rewards/meter/mean": 0.5907192230224609, "rewards/meter/std": 0.3661164939403534, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.4524690806865692, "rewards/total_composite/std": 0.37379953265190125, "reward": 0.4524690806865692, "reward_std": 0.37379953265190125, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14518797397613525, "sampling/sampling_logp_difference/max": 1.4015388488769531, "sampling/importance_sampling_ratio/min": 0.2462177872657776, "sampling/importance_sampling_ratio/mean": 1.0290663242340088, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7579984441399574, "clip_ratio/low_mean": 0.04759177938103676, "clip_ratio/low_min": 0.04759177938103676, "clip_ratio/high_mean": 0.050237943418323994, "clip_ratio/high_max": 0.050237943418323994, "clip_ratio/region_mean": 0.09782972279936075, "reward_total_mean": 0.4524690806865692, "reward_meter_mean": 0.5907192230224609, "reward_meter_std": 0.3661164939403534, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.4524690806865692, "reward_total_composite_std": 0.37379953265190125, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 292.0} {"timestamp_utc": "2026-04-11T19:57:33Z", "mode": "train", "global_step": 293, "epoch": 0.011314488724127278, "loss": 0.0185, "grad_norm": 9.675642967224121, "learning_rate": 9.115151515151516e-06, "num_tokens": 632410.0, "completions/mean_length": 64.25, "completions/min_length": 54.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9181022047996521, "rewards/meter/std": 0.09518402069807053, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9181022047996521, "rewards/total_composite/std": 0.09518402069807053, "reward": 0.9181022047996521, "reward_std": 0.09518400579690933, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15963061153888702, "sampling/sampling_logp_difference/max": 1.8079185485839844, "sampling/importance_sampling_ratio/min": 0.16399513185024261, "sampling/importance_sampling_ratio/mean": 1.036954402923584, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.85820072889328, "clip_ratio/low_mean": 0.051651342771947384, "clip_ratio/low_min": 0.051651342771947384, "clip_ratio/high_mean": 0.10475165769457817, "clip_ratio/high_max": 0.10475165769457817, "clip_ratio/region_mean": 0.15640300046652555, "reward_total_mean": 0.9181022047996521, "reward_meter_mean": 0.9181022047996521, "reward_meter_std": 0.09518402069807053, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9181022047996521, "reward_total_composite_std": 0.09518402069807053, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 293.0} {"timestamp_utc": "2026-04-11T19:57:38Z", "mode": "train", "global_step": 294, "epoch": 0.011353104726598702, "loss": -0.022, "grad_norm": 22.611948013305664, "learning_rate": 9.112121212121214e-06, "num_tokens": 633858.0, "completions/mean_length": 31.0, "completions/min_length": 25.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.8932417631149292, "rewards/meter/std": 0.2514517903327942, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8932417631149292, "rewards/total_composite/std": 0.2514517903327942, "reward": 0.8932417631149292, "reward_std": 0.2514517605304718, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18085245788097382, "sampling/sampling_logp_difference/max": 2.2035961151123047, "sampling/importance_sampling_ratio/min": 0.11040541529655457, "sampling/importance_sampling_ratio/mean": 1.0493816137313843, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2211313992738724, "clip_ratio/low_mean": 0.008928571827709675, "clip_ratio/low_min": 0.008928571827709675, "clip_ratio/high_mean": 0.15911428444087505, "clip_ratio/high_max": 0.15911428444087505, "clip_ratio/region_mean": 0.16804285626858473, "reward_total_mean": 0.8932417631149292, "reward_meter_mean": 0.8932417631149292, "reward_meter_std": 0.2514517903327942, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8932417631149292, "reward_total_composite_std": 0.2514517903327942, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 294.0} {"timestamp_utc": "2026-04-11T19:57:43Z", "mode": "train", "global_step": 295, "epoch": 0.011391720729070126, "loss": 0.05, "grad_norm": 11.888802528381348, "learning_rate": 9.10909090909091e-06, "num_tokens": 635653.0, "completions/mean_length": 63.375, "completions/min_length": 58.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.375, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7532914876937866, "rewards/meter/std": 0.3347664475440979, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7532914876937866, "rewards/total_composite/std": 0.3347664475440979, "reward": 0.7532914876937866, "reward_std": 0.3347664773464203, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1385606974363327, "sampling/sampling_logp_difference/max": 1.5744085311889648, "sampling/importance_sampling_ratio/min": 0.20713002979755402, "sampling/importance_sampling_ratio/mean": 1.0173488855361938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5011917799711227, "clip_ratio/low_mean": 0.05079313600435853, "clip_ratio/low_min": 0.05079313600435853, "clip_ratio/high_mean": 0.08510276302695274, "clip_ratio/high_max": 0.08510276302695274, "clip_ratio/region_mean": 0.13589589903131127, "reward_total_mean": 0.7532914876937866, "reward_meter_mean": 0.7532914876937866, "reward_meter_std": 0.3347664475440979, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7532914876937866, "reward_total_composite_std": 0.3347664475440979, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 295.0} {"timestamp_utc": "2026-04-11T19:57:47Z", "mode": "train", "global_step": 296, "epoch": 0.01143033673154155, "loss": 0.1775, "grad_norm": 37.00153732299805, "learning_rate": 9.106060606060606e-06, "num_tokens": 637188.0, "completions/mean_length": 25.875, "completions/min_length": 20.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.875, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8021076321601868, "rewards/meter/std": 0.21438796818256378, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8021076321601868, "rewards/total_composite/std": 0.21438796818256378, "reward": 0.8021076321601868, "reward_std": 0.21438796818256378, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15415281057357788, "sampling/sampling_logp_difference/max": 2.5325937271118164, "sampling/importance_sampling_ratio/min": 0.07945267111063004, "sampling/importance_sampling_ratio/mean": 0.9893893003463745, "sampling/importance_sampling_ratio/max": 1.9617925882339478, "entropy": 0.7283558156341314, "clip_ratio/low_mean": 0.06392320711165667, "clip_ratio/low_min": 0.06392320711165667, "clip_ratio/high_mean": 0.054404761642217636, "clip_ratio/high_max": 0.054404761642217636, "clip_ratio/region_mean": 0.1183279687538743, "reward_total_mean": 0.8021076321601868, "reward_meter_mean": 0.8021076321601868, "reward_meter_std": 0.21438796818256378, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8021076321601868, "reward_total_composite_std": 0.21438796818256378, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 296.0} {"timestamp_utc": "2026-04-11T19:57:52Z", "mode": "train", "global_step": 297, "epoch": 0.011468952734012975, "loss": 0.0003, "grad_norm": 10.021339416503906, "learning_rate": 9.103030303030304e-06, "num_tokens": 639349.0, "completions/mean_length": 90.125, "completions/min_length": 81.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.125, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.6162939667701721, "rewards/meter/std": 0.46589404344558716, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6162939667701721, "rewards/total_composite/std": 0.46589404344558716, "reward": 0.6162939667701721, "reward_std": 0.46589401364326477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13409848511219025, "sampling/sampling_logp_difference/max": 1.4236335754394531, "sampling/importance_sampling_ratio/min": 0.24083733558654785, "sampling/importance_sampling_ratio/mean": 1.0367764234542847, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3437079191207886, "clip_ratio/low_mean": 0.046736557967960835, "clip_ratio/low_min": 0.046736557967960835, "clip_ratio/high_mean": 0.06547143869102001, "clip_ratio/high_max": 0.06547143869102001, "clip_ratio/region_mean": 0.11220799665898085, "reward_total_mean": 0.6162939667701721, "reward_meter_mean": 0.6162939667701721, "reward_meter_std": 0.46589404344558716, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6162939667701721, "reward_total_composite_std": 0.46589404344558716, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 297.0} {"timestamp_utc": "2026-04-11T19:58:00Z", "mode": "train", "global_step": 298, "epoch": 0.011507568736484399, "loss": 0.0128, "grad_norm": 6.047693729400635, "learning_rate": 9.100000000000001e-06, "num_tokens": 642375.0, "completions/mean_length": 183.25, "completions/min_length": 151.0, "completions/max_length": 212.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 183.25, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 212.0, "rewards/meter/mean": 0.9324739575386047, "rewards/meter/std": 0.09105537831783295, "rewards/count_adherence/mean": 0.8214285373687744, "rewards/count_adherence/std": 0.06613000482320786, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7657866477966309, "rewards/total_composite/std": 0.09787718951702118, "reward": 0.7657866477966309, "reward_std": 0.09787716716527939, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14809899032115936, "sampling/sampling_logp_difference/max": 3.5092623233795166, "sampling/importance_sampling_ratio/min": 0.02991897612810135, "sampling/importance_sampling_ratio/mean": 1.0216869115829468, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7826667726039886, "clip_ratio/low_mean": 0.03513659443706274, "clip_ratio/low_min": 0.03513659443706274, "clip_ratio/high_mean": 0.0976903848350048, "clip_ratio/high_max": 0.0976903848350048, "clip_ratio/region_mean": 0.13282697927206755, "reward_total_mean": 0.7657866477966309, "reward_meter_mean": 0.9324739575386047, "reward_meter_std": 0.09105537831783295, "reward_count_adherence_mean": 0.8214285373687744, "reward_count_adherence_std": 0.06613000482320786, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7657866477966309, "reward_total_composite_std": 0.09787718951702118, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 298.0} {"timestamp_utc": "2026-04-11T19:58:05Z", "mode": "train", "global_step": 299, "epoch": 0.011546184738955823, "loss": 0.1033, "grad_norm": 12.910846710205078, "learning_rate": 9.096969696969698e-06, "num_tokens": 644161.0, "completions/mean_length": 65.25, "completions/min_length": 57.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.25, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.8054907917976379, "rewards/meter/std": 0.36134690046310425, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7445110082626343, "rewards/total_composite/std": 0.3695930540561676, "reward": 0.7445110082626343, "reward_std": 0.3695930242538452, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16549576818943024, "sampling/sampling_logp_difference/max": 1.338456153869629, "sampling/importance_sampling_ratio/min": 0.26225021481513977, "sampling/importance_sampling_ratio/mean": 1.0238062143325806, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.74993297457695, "clip_ratio/low_mean": 0.046733343973755836, "clip_ratio/low_min": 0.046733343973755836, "clip_ratio/high_mean": 0.10818896908313036, "clip_ratio/high_max": 0.10818896908313036, "clip_ratio/region_mean": 0.1549223130568862, "reward_total_mean": 0.7445110082626343, "reward_meter_mean": 0.8054907917976379, "reward_meter_std": 0.36134690046310425, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7445110082626343, "reward_total_composite_std": 0.3695930540561676, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 299.0} {"timestamp_utc": "2026-04-11T19:58:09Z", "mode": "train", "global_step": 300, "epoch": 0.011584800741427247, "loss": 0.0899, "grad_norm": 13.964972496032715, "learning_rate": 9.093939393939395e-06, "num_tokens": 645560.0, "completions/mean_length": 28.875, "completions/min_length": 25.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6676859259605408, "rewards/meter/std": 0.40630266070365906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6676859259605408, "rewards/total_composite/std": 0.40630266070365906, "reward": 0.6676859259605408, "reward_std": 0.40630266070365906, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17699161171913147, "sampling/sampling_logp_difference/max": 1.2114429473876953, "sampling/importance_sampling_ratio/min": 0.29776731133461, "sampling/importance_sampling_ratio/mean": 1.0266624689102173, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.053573176264763, "clip_ratio/low_mean": 0.06600985117256641, "clip_ratio/low_min": 0.06600985117256641, "clip_ratio/high_mean": 0.09760686475783587, "clip_ratio/high_max": 0.09760686475783587, "clip_ratio/region_mean": 0.16361671593040228, "reward_total_mean": 0.6676859259605408, "reward_meter_mean": 0.6676859259605408, "reward_meter_std": 0.40630266070365906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6676859259605408, "reward_total_composite_std": 0.40630266070365906, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 300.0} {"timestamp_utc": "2026-04-11T19:59:26Z", "mode": "eval", "global_step": 300, "epoch": 0.011584800741427247, "eval_loss": NaN, "eval_runtime": 76.9936, "eval_samples_per_second": 1.351, "eval_steps_per_second": 0.169, "eval_num_tokens": 645560.0, "eval_completions/mean_length": 189.27884615384616, "eval_completions/min_length": 48.69230769230769, "eval_completions/max_length": 413.84615384615387, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/mean_terminated_length": 169.91621281550482, "eval_completions/min_terminated_length": 48.69230769230769, "eval_completions/max_terminated_length": 332.6923076923077, "eval_rewards/meter/mean": 0.5671234520582052, "eval_rewards/meter/std": 0.3898524023019351, "eval_rewards/count_adherence/mean": 0.8583941322106582, "eval_rewards/count_adherence/std": 0.16477382297699267, "eval_rewards/arabic_clean/mean": 0.9326923076923077, "eval_rewards/arabic_clean/std": 0.19037489936901972, "eval_rewards/total_composite/mean": 0.47910642050779784, "eval_rewards/total_composite/std": 0.3604818215736976, "eval_reward": 0.47910642050779784, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.10725439225251858, "eval_sampling/sampling_logp_difference/max": 1.1390617810762846, "eval_sampling/importance_sampling_ratio/min": 0.32239050360826343, "eval_sampling/importance_sampling_ratio/mean": 1.0292121997246375, "eval_sampling/importance_sampling_ratio/max": 1.5635485924207246, "eval_entropy": 1.6832339671941905, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.47910642050779784, "eval_reward_meter_mean": 0.5671234520582052, "eval_reward_meter_std": 0.3898524023019351, "eval_reward_count_adherence_mean": 0.8583941322106582, "eval_reward_count_adherence_std": 0.16477382297699267, "eval_reward_arabic_clean_mean": 0.9326923076923077, "eval_reward_arabic_clean_std": 0.19037489936901972, "eval_reward_total_composite_mean": 0.47910642050779784, "eval_reward_total_composite_std": 0.3604818215736976, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 300.0} {"timestamp_utc": "2026-04-11T19:59:39Z", "mode": "train", "global_step": 301, "epoch": 0.011623416743898671, "loss": -0.313, "grad_norm": 1.71794855594635, "learning_rate": 9.090909090909091e-06, "num_tokens": 648373.0, "completions/mean_length": 462.625, "completions/min_length": 205.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 380.3333435058594, "completions/min_terminated_length": 205.0, "completions/max_terminated_length": 471.0, "rewards/meter/mean": 0.6754355430603027, "rewards/meter/std": 0.3314078450202942, "rewards/count_adherence/mean": 0.4375, "rewards/count_adherence/std": 0.18169663846492767, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/total_composite/mean": 0.21318131685256958, "rewards/total_composite/std": 0.2553112804889679, "reward": 0.21318131685256958, "reward_std": 0.2553112506866455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13692021369934082, "sampling/sampling_logp_difference/max": 1.3053817749023438, "sampling/importance_sampling_ratio/min": 0.2710690200328827, "sampling/importance_sampling_ratio/mean": 1.0243096351623535, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7612107247114182, "clip_ratio/low_mean": 0.012804877944290638, "clip_ratio/low_min": 0.012804877944290638, "clip_ratio/high_mean": 0.021113280206918716, "clip_ratio/high_max": 0.021113280206918716, "clip_ratio/region_mean": 0.033918158151209354, "reward_total_mean": 0.21318131685256958, "reward_meter_mean": 0.6754355430603027, "reward_meter_std": 0.3314078450202942, "reward_count_adherence_mean": 0.4375, "reward_count_adherence_std": 0.18169663846492767, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_total_composite_mean": 0.21318131685256958, "reward_total_composite_std": 0.2553112804889679, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 301.0} {"timestamp_utc": "2026-04-11T19:59:45Z", "mode": "train", "global_step": 302, "epoch": 0.011662032746370095, "loss": -0.0483, "grad_norm": 6.904706954956055, "learning_rate": 9.087878787878788e-06, "num_tokens": 651028.0, "completions/mean_length": 131.875, "completions/min_length": 117.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.875, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.407183438539505, "rewards/meter/std": 0.17483185231685638, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3257467448711395, "rewards/total_composite/std": 0.1398654729127884, "reward": 0.3257467448711395, "reward_std": 0.1398654729127884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17297177016735077, "sampling/sampling_logp_difference/max": 2.081483840942383, "sampling/importance_sampling_ratio/min": 0.20558007061481476, "sampling/importance_sampling_ratio/mean": 1.031490445137024, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9741446822881699, "clip_ratio/low_mean": 0.06732289679348469, "clip_ratio/low_min": 0.06732289679348469, "clip_ratio/high_mean": 0.10167329479008913, "clip_ratio/high_max": 0.10167329479008913, "clip_ratio/region_mean": 0.16899619158357382, "reward_total_mean": 0.3257467448711395, "reward_meter_mean": 0.407183438539505, "reward_meter_std": 0.17483185231685638, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3257467448711395, "reward_total_composite_std": 0.1398654729127884, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 302.0} {"timestamp_utc": "2026-04-11T19:59:49Z", "mode": "train", "global_step": 303, "epoch": 0.01170064874884152, "loss": -0.0003, "grad_norm": 13.510102272033691, "learning_rate": 9.084848484848486e-06, "num_tokens": 652679.0, "completions/mean_length": 35.375, "completions/min_length": 32.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.6417617797851562, "rewards/meter/std": 0.47610655426979065, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6417617797851562, "rewards/total_composite/std": 0.47610655426979065, "reward": 0.6417617797851562, "reward_std": 0.47610652446746826, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16745901107788086, "sampling/sampling_logp_difference/max": 1.8051767349243164, "sampling/importance_sampling_ratio/min": 0.16444538533687592, "sampling/importance_sampling_ratio/mean": 1.001904010772705, "sampling/importance_sampling_ratio/max": 1.849202036857605, "entropy": 1.5965645760297775, "clip_ratio/low_mean": 0.03281614277511835, "clip_ratio/low_min": 0.03281614277511835, "clip_ratio/high_mean": 0.0840634061023593, "clip_ratio/high_max": 0.0840634061023593, "clip_ratio/region_mean": 0.11687954887747765, "reward_total_mean": 0.6417617797851562, "reward_meter_mean": 0.6417617797851562, "reward_meter_std": 0.47610655426979065, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6417617797851562, "reward_total_composite_std": 0.47610655426979065, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 303.0} {"timestamp_utc": "2026-04-11T19:59:54Z", "mode": "train", "global_step": 304, "epoch": 0.011739264751312943, "loss": -0.0536, "grad_norm": 9.380762100219727, "learning_rate": 9.081818181818183e-06, "num_tokens": 654531.0, "completions/mean_length": 65.5, "completions/min_length": 50.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.6257134079933167, "rewards/meter/std": 0.39615458250045776, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5989729166030884, "rewards/total_composite/std": 0.43339118361473083, "reward": 0.5989729166030884, "reward_std": 0.43339118361473083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.174082413315773, "sampling/sampling_logp_difference/max": 1.9137487411499023, "sampling/importance_sampling_ratio/min": 0.1475263237953186, "sampling/importance_sampling_ratio/mean": 1.0195375680923462, "sampling/importance_sampling_ratio/max": 1.9768328666687012, "entropy": 1.989267036318779, "clip_ratio/low_mean": 0.033333334140479565, "clip_ratio/low_min": 0.033333334140479565, "clip_ratio/high_mean": 0.08884815126657486, "clip_ratio/high_max": 0.08884815126657486, "clip_ratio/region_mean": 0.12218148540705442, "reward_total_mean": 0.5989729166030884, "reward_meter_mean": 0.6257134079933167, "reward_meter_std": 0.39615458250045776, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5989729166030884, "reward_total_composite_std": 0.43339118361473083, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 304.0} {"timestamp_utc": "2026-04-11T19:59:59Z", "mode": "train", "global_step": 305, "epoch": 0.011777880753784368, "loss": 0.0498, "grad_norm": 10.915406227111816, "learning_rate": 9.078787878787878e-06, "num_tokens": 656283.0, "completions/mean_length": 63.0, "completions/min_length": 55.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8347302079200745, "rewards/meter/std": 0.3262871503829956, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8347302079200745, "rewards/total_composite/std": 0.3262871503829956, "reward": 0.8347302079200745, "reward_std": 0.3262871503829956, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1379973143339157, "sampling/sampling_logp_difference/max": 1.5258607864379883, "sampling/importance_sampling_ratio/min": 0.21743382513523102, "sampling/importance_sampling_ratio/mean": 1.0057920217514038, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5056721195578575, "clip_ratio/low_mean": 0.027922078035771847, "clip_ratio/low_min": 0.027922078035771847, "clip_ratio/high_mean": 0.10270242113620043, "clip_ratio/high_max": 0.10270242113620043, "clip_ratio/region_mean": 0.13062449917197227, "reward_total_mean": 0.8347302079200745, "reward_meter_mean": 0.8347302079200745, "reward_meter_std": 0.3262871503829956, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8347302079200745, "reward_total_composite_std": 0.3262871503829956, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 305.0} {"timestamp_utc": "2026-04-11T20:00:03Z", "mode": "train", "global_step": 306, "epoch": 0.011816496756255792, "loss": -0.0637, "grad_norm": 15.107892036437988, "learning_rate": 9.075757575757577e-06, "num_tokens": 657681.0, "completions/mean_length": 21.75, "completions/min_length": 15.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.75, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.6698324680328369, "rewards/meter/std": 0.42919304966926575, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6698324680328369, "rewards/total_composite/std": 0.42919304966926575, "reward": 0.6698324680328369, "reward_std": 0.42919301986694336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18782852590084076, "sampling/sampling_logp_difference/max": 1.1668949127197266, "sampling/importance_sampling_ratio/min": 0.3113321363925934, "sampling/importance_sampling_ratio/mean": 1.038679838180542, "sampling/importance_sampling_ratio/max": 1.7879084348678589, "entropy": 2.3475464656949043, "clip_ratio/low_mean": 0.09684343822300434, "clip_ratio/low_min": 0.09684343822300434, "clip_ratio/high_mean": 0.10306308604776859, "clip_ratio/high_max": 0.10306308604776859, "clip_ratio/region_mean": 0.19990652427077293, "reward_total_mean": 0.6698324680328369, "reward_meter_mean": 0.6698324680328369, "reward_meter_std": 0.42919304966926575, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6698324680328369, "reward_total_composite_std": 0.42919304966926575, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 306.0} {"timestamp_utc": "2026-04-11T20:00:08Z", "mode": "train", "global_step": 307, "epoch": 0.011855112758727216, "loss": 0.0343, "grad_norm": 7.24405574798584, "learning_rate": 9.072727272727273e-06, "num_tokens": 659813.0, "completions/mean_length": 103.5, "completions/min_length": 88.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.5, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.5797778367996216, "rewards/meter/std": 0.4139650762081146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.4702509343624115, "rewards/total_composite/std": 0.4394587278366089, "reward": 0.4702509343624115, "reward_std": 0.4394586980342865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16522862017154694, "sampling/sampling_logp_difference/max": 1.6513099670410156, "sampling/importance_sampling_ratio/min": 0.19179849326610565, "sampling/importance_sampling_ratio/mean": 1.0307226181030273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.192972496151924, "clip_ratio/low_mean": 0.08673457149416208, "clip_ratio/low_min": 0.08673457149416208, "clip_ratio/high_mean": 0.064244344830513, "clip_ratio/high_max": 0.064244344830513, "clip_ratio/region_mean": 0.15097891632467508, "reward_total_mean": 0.4702509343624115, "reward_meter_mean": 0.5797778367996216, "reward_meter_std": 0.4139650762081146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.4702509343624115, "reward_total_composite_std": 0.4394587278366089, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 307.0} {"timestamp_utc": "2026-04-11T20:00:13Z", "mode": "train", "global_step": 308, "epoch": 0.011893728761198642, "loss": -0.0505, "grad_norm": 12.200899124145508, "learning_rate": 9.06969696969697e-06, "num_tokens": 661897.0, "completions/mean_length": 86.5, "completions/min_length": 67.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.5, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.5630761384963989, "rewards/meter/std": 0.37055954337120056, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5569584369659424, "rewards/total_composite/std": 0.3778226375579834, "reward": 0.5569584369659424, "reward_std": 0.3778226375579834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16500020027160645, "sampling/sampling_logp_difference/max": 2.038670539855957, "sampling/importance_sampling_ratio/min": 0.13020169734954834, "sampling/importance_sampling_ratio/mean": 1.0206682682037354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4088008925318718, "clip_ratio/low_mean": 0.07120703207328916, "clip_ratio/low_min": 0.07120703207328916, "clip_ratio/high_mean": 0.05581671930849552, "clip_ratio/high_max": 0.05581671930849552, "clip_ratio/region_mean": 0.12702375138178468, "reward_total_mean": 0.5569584369659424, "reward_meter_mean": 0.5630761384963989, "reward_meter_std": 0.37055954337120056, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5569584369659424, "reward_total_composite_std": 0.3778226375579834, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 308.0} {"timestamp_utc": "2026-04-11T20:00:18Z", "mode": "train", "global_step": 309, "epoch": 0.011932344763670066, "loss": -0.0083, "grad_norm": 9.461854934692383, "learning_rate": 9.066666666666667e-06, "num_tokens": 663904.0, "completions/mean_length": 84.875, "completions/min_length": 55.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.875, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.6280455589294434, "rewards/meter/std": 0.17267122864723206, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.607955813407898, "rewards/total_composite/std": 0.19935742020606995, "reward": 0.607955813407898, "reward_std": 0.19935740530490875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17172668874263763, "sampling/sampling_logp_difference/max": 1.3207921981811523, "sampling/importance_sampling_ratio/min": 0.2669237554073334, "sampling/importance_sampling_ratio/mean": 1.0375332832336426, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.261012151837349, "clip_ratio/low_mean": 0.06721637584269047, "clip_ratio/low_min": 0.06721637584269047, "clip_ratio/high_mean": 0.08292348962277174, "clip_ratio/high_max": 0.08292348962277174, "clip_ratio/region_mean": 0.1501398654654622, "reward_total_mean": 0.607955813407898, "reward_meter_mean": 0.6280455589294434, "reward_meter_std": 0.17267122864723206, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.607955813407898, "reward_total_composite_std": 0.19935742020606995, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 309.0} {"timestamp_utc": "2026-04-11T20:00:24Z", "mode": "train", "global_step": 310, "epoch": 0.01197096076614149, "loss": -0.0331, "grad_norm": 6.23518705368042, "learning_rate": 9.063636363636365e-06, "num_tokens": 666273.0, "completions/mean_length": 129.125, "completions/min_length": 111.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.125, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.7603760957717896, "rewards/meter/std": 0.25764790177345276, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6215674877166748, "rewards/total_composite/std": 0.19575539231300354, "reward": 0.6215674877166748, "reward_std": 0.19575539231300354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14828266203403473, "sampling/sampling_logp_difference/max": 1.5174407958984375, "sampling/importance_sampling_ratio/min": 0.21927233040332794, "sampling/importance_sampling_ratio/mean": 1.0316812992095947, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.74216927587986, "clip_ratio/low_mean": 0.061003027483820915, "clip_ratio/low_min": 0.061003027483820915, "clip_ratio/high_mean": 0.06727708876132965, "clip_ratio/high_max": 0.06727708876132965, "clip_ratio/region_mean": 0.12828011624515057, "reward_total_mean": 0.6215674877166748, "reward_meter_mean": 0.7603760957717896, "reward_meter_std": 0.25764790177345276, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6215674877166748, "reward_total_composite_std": 0.19575539231300354, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 310.0} {"timestamp_utc": "2026-04-11T20:00:30Z", "mode": "train", "global_step": 311, "epoch": 0.012009576768612914, "loss": 0.1673, "grad_norm": 15.047697067260742, "learning_rate": 9.06060606060606e-06, "num_tokens": 668998.0, "completions/mean_length": 141.625, "completions/min_length": 103.0, "completions/max_length": 203.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.625, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 203.0, "rewards/meter/mean": 0.7493094205856323, "rewards/meter/std": 0.22875480353832245, "rewards/count_adherence/mean": 0.953125, "rewards/count_adherence/std": 0.06469365209341049, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7146421670913696, "rewards/total_composite/std": 0.2316495180130005, "reward": 0.7146421670913696, "reward_std": 0.2316495180130005, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21245455741882324, "sampling/sampling_logp_difference/max": 3.1872100830078125, "sampling/importance_sampling_ratio/min": 0.041286900639534, "sampling/importance_sampling_ratio/mean": 0.9966016411781311, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3266419470310211, "clip_ratio/low_mean": 0.06916317716240883, "clip_ratio/low_min": 0.06916317716240883, "clip_ratio/high_mean": 0.1306404136121273, "clip_ratio/high_max": 0.1306404136121273, "clip_ratio/region_mean": 0.19980359077453613, "reward_total_mean": 0.7146421670913696, "reward_meter_mean": 0.7493094205856323, "reward_meter_std": 0.22875480353832245, "reward_count_adherence_mean": 0.953125, "reward_count_adherence_std": 0.06469365209341049, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7146421670913696, "reward_total_composite_std": 0.2316495180130005, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 311.0} {"timestamp_utc": "2026-04-11T20:00:34Z", "mode": "train", "global_step": 312, "epoch": 0.012048192771084338, "loss": -0.011, "grad_norm": 19.782384872436523, "learning_rate": 9.057575757575759e-06, "num_tokens": 670353.0, "completions/mean_length": 30.375, "completions/min_length": 26.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.375, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.7651474475860596, "rewards/meter/std": 0.3580241799354553, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7651474475860596, "rewards/total_composite/std": 0.3580241799354553, "reward": 0.7651474475860596, "reward_std": 0.35802415013313293, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16778768599033356, "sampling/sampling_logp_difference/max": 1.0911779403686523, "sampling/importance_sampling_ratio/min": 0.33582067489624023, "sampling/importance_sampling_ratio/mean": 1.0236084461212158, "sampling/importance_sampling_ratio/max": 1.8821232318878174, "entropy": 1.9258029609918594, "clip_ratio/low_mean": 0.030448718927800655, "clip_ratio/low_min": 0.030448718927800655, "clip_ratio/high_mean": 0.15273912716656923, "clip_ratio/high_max": 0.15273912716656923, "clip_ratio/region_mean": 0.1831878460943699, "reward_total_mean": 0.7651474475860596, "reward_meter_mean": 0.7651474475860596, "reward_meter_std": 0.3580241799354553, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7651474475860596, "reward_total_composite_std": 0.3580241799354553, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 312.0} {"timestamp_utc": "2026-04-11T20:00:38Z", "mode": "train", "global_step": 313, "epoch": 0.012086808773555762, "loss": -0.0656, "grad_norm": 10.562493324279785, "learning_rate": 9.054545454545455e-06, "num_tokens": 671977.0, "completions/mean_length": 62.0, "completions/min_length": 50.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9314479231834412, "rewards/meter/std": 0.10455489158630371, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9314479231834412, "rewards/total_composite/std": 0.10455489158630371, "reward": 0.9314479231834412, "reward_std": 0.10455489158630371, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1677490621805191, "sampling/sampling_logp_difference/max": 1.466294288635254, "sampling/importance_sampling_ratio/min": 0.23077911138534546, "sampling/importance_sampling_ratio/mean": 1.0549638271331787, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.047884911298752, "clip_ratio/low_mean": 0.03583333361893892, "clip_ratio/low_min": 0.03583333361893892, "clip_ratio/high_mean": 0.10207792790606618, "clip_ratio/high_max": 0.10207792790606618, "clip_ratio/region_mean": 0.1379112615250051, "reward_total_mean": 0.9314479231834412, "reward_meter_mean": 0.9314479231834412, "reward_meter_std": 0.10455489158630371, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9314479231834412, "reward_total_composite_std": 0.10455489158630371, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 313.0} {"timestamp_utc": "2026-04-11T20:00:43Z", "mode": "train", "global_step": 314, "epoch": 0.012125424776027186, "loss": -0.0042, "grad_norm": 16.10173797607422, "learning_rate": 9.051515151515152e-06, "num_tokens": 673690.0, "completions/mean_length": 62.125, "completions/min_length": 56.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8384339809417725, "rewards/meter/std": 0.33158746361732483, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8384339809417725, "rewards/total_composite/std": 0.33158746361732483, "reward": 0.8384339809417725, "reward_std": 0.33158746361732483, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15842776000499725, "sampling/sampling_logp_difference/max": 1.6559438705444336, "sampling/importance_sampling_ratio/min": 0.19091176986694336, "sampling/importance_sampling_ratio/mean": 1.0274717807769775, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8069640547037125, "clip_ratio/low_mean": 0.02873883955180645, "clip_ratio/low_min": 0.02873883955180645, "clip_ratio/high_mean": 0.12252288311719894, "clip_ratio/high_max": 0.12252288311719894, "clip_ratio/region_mean": 0.1512617226690054, "reward_total_mean": 0.8384339809417725, "reward_meter_mean": 0.8384339809417725, "reward_meter_std": 0.33158746361732483, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8384339809417725, "reward_total_composite_std": 0.33158746361732483, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 314.0} {"timestamp_utc": "2026-04-11T20:00:48Z", "mode": "train", "global_step": 315, "epoch": 0.01216404077849861, "loss": -0.01, "grad_norm": 6.204983234405518, "learning_rate": 9.04848484848485e-06, "num_tokens": 676000.0, "completions/mean_length": 111.75, "completions/min_length": 103.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.75, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.9871163368225098, "rewards/meter/std": 0.022878479212522507, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9871163368225098, "rewards/total_composite/std": 0.022878479212522507, "reward": 0.9871163368225098, "reward_std": 0.022878482937812805, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13204924762248993, "sampling/sampling_logp_difference/max": 1.2926632165908813, "sampling/importance_sampling_ratio/min": 0.27453866600990295, "sampling/importance_sampling_ratio/mean": 1.0214686393737793, "sampling/importance_sampling_ratio/max": 1.6907645463943481, "entropy": 1.6218044012784958, "clip_ratio/low_mean": 0.016509434208273888, "clip_ratio/low_min": 0.016509434208273888, "clip_ratio/high_mean": 0.09765755478292704, "clip_ratio/high_max": 0.09765755478292704, "clip_ratio/region_mean": 0.11416698899120092, "reward_total_mean": 0.9871163368225098, "reward_meter_mean": 0.9871163368225098, "reward_meter_std": 0.022878479212522507, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9871163368225098, "reward_total_composite_std": 0.022878479212522507, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 315.0} {"timestamp_utc": "2026-04-11T20:00:54Z", "mode": "train", "global_step": 316, "epoch": 0.012202656780970034, "loss": 0.0004, "grad_norm": 6.30336332321167, "learning_rate": 9.045454545454546e-06, "num_tokens": 678856.0, "completions/mean_length": 161.0, "completions/min_length": 153.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 161.0, "completions/min_terminated_length": 153.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.412434846162796, "rewards/meter/std": 0.3995774984359741, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.343695729970932, "rewards/total_composite/std": 0.33298125863075256, "reward": 0.343695729970932, "reward_std": 0.3329812288284302, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14012373983860016, "sampling/sampling_logp_difference/max": 1.7056856155395508, "sampling/importance_sampling_ratio/min": 0.18164780735969543, "sampling/importance_sampling_ratio/mean": 1.0196648836135864, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.385835349559784, "clip_ratio/low_mean": 0.0671336529776454, "clip_ratio/low_min": 0.0671336529776454, "clip_ratio/high_mean": 0.0503198578953743, "clip_ratio/high_max": 0.0503198578953743, "clip_ratio/region_mean": 0.1174535108730197, "reward_total_mean": 0.343695729970932, "reward_meter_mean": 0.412434846162796, "reward_meter_std": 0.3995774984359741, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.343695729970932, "reward_total_composite_std": 0.33298125863075256, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 316.0} {"timestamp_utc": "2026-04-11T20:00:59Z", "mode": "train", "global_step": 317, "epoch": 0.012241272783441458, "loss": -0.0392, "grad_norm": 11.422844886779785, "learning_rate": 9.042424242424244e-06, "num_tokens": 680465.0, "completions/mean_length": 59.125, "completions/min_length": 46.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9061346650123596, "rewards/meter/std": 0.22870448231697083, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9061346650123596, "rewards/total_composite/std": 0.22870448231697083, "reward": 0.9061346650123596, "reward_std": 0.22870448231697083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14648805558681488, "sampling/sampling_logp_difference/max": 1.214674472808838, "sampling/importance_sampling_ratio/min": 0.29680660367012024, "sampling/importance_sampling_ratio/mean": 1.0229982137680054, "sampling/importance_sampling_ratio/max": 1.9018646478652954, "entropy": 1.7790979892015457, "clip_ratio/low_mean": 0.020408162847161293, "clip_ratio/low_min": 0.020408162847161293, "clip_ratio/high_mean": 0.13084796536713839, "clip_ratio/high_max": 0.13084796536713839, "clip_ratio/region_mean": 0.15125612821429968, "reward_total_mean": 0.9061346650123596, "reward_meter_mean": 0.9061346650123596, "reward_meter_std": 0.22870448231697083, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9061346650123596, "reward_total_composite_std": 0.22870448231697083, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 317.0} {"timestamp_utc": "2026-04-11T20:01:03Z", "mode": "train", "global_step": 318, "epoch": 0.012279888785912883, "loss": -0.0306, "grad_norm": 8.7025728225708, "learning_rate": 9.03939393939394e-06, "num_tokens": 682172.0, "completions/mean_length": 70.375, "completions/min_length": 55.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9199281930923462, "rewards/meter/std": 0.17555859684944153, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9199281930923462, "rewards/total_composite/std": 0.17555859684944153, "reward": 0.9199281930923462, "reward_std": 0.17555858194828033, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1701533943414688, "sampling/sampling_logp_difference/max": 1.4907312393188477, "sampling/importance_sampling_ratio/min": 0.22520792484283447, "sampling/importance_sampling_ratio/mean": 1.0345814228057861, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0886534601449966, "clip_ratio/low_mean": 0.015384615398943424, "clip_ratio/low_min": 0.015384615398943424, "clip_ratio/high_mean": 0.13356309290975332, "clip_ratio/high_max": 0.13356309290975332, "clip_ratio/region_mean": 0.14894770830869675, "reward_total_mean": 0.9199281930923462, "reward_meter_mean": 0.9199281930923462, "reward_meter_std": 0.17555859684944153, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9199281930923462, "reward_total_composite_std": 0.17555859684944153, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 318.0} {"timestamp_utc": "2026-04-11T20:01:13Z", "mode": "train", "global_step": 319, "epoch": 0.012318504788384307, "loss": -0.1669, "grad_norm": 2.1287405490875244, "learning_rate": 9.036363636363638e-06, "num_tokens": 684570.0, "completions/mean_length": 309.75, "completions/min_length": 163.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 188.40000915527344, "completions/min_terminated_length": 163.0, "completions/max_terminated_length": 213.0, "rewards/meter/mean": 0.6731476783752441, "rewards/meter/std": 0.4353531301021576, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.14880475401878357, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6160376071929932, "rewards/total_composite/std": 0.4165080785751343, "reward": 0.6160376071929932, "reward_std": 0.4165080487728119, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11113591492176056, "sampling/sampling_logp_difference/max": 1.4681377410888672, "sampling/importance_sampling_ratio/min": 0.23035407066345215, "sampling/importance_sampling_ratio/mean": 1.0229874849319458, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8439691588282585, "clip_ratio/low_mean": 0.014423076994717121, "clip_ratio/low_min": 0.014423076994717121, "clip_ratio/high_mean": 0.035777142737060785, "clip_ratio/high_max": 0.035777142737060785, "clip_ratio/region_mean": 0.050200219731777906, "reward_total_mean": 0.6160376071929932, "reward_meter_mean": 0.6731476783752441, "reward_meter_std": 0.4353531301021576, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.14880475401878357, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6160376071929932, "reward_total_composite_std": 0.4165080785751343, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 319.0} {"timestamp_utc": "2026-04-11T20:01:18Z", "mode": "train", "global_step": 320, "epoch": 0.01235712079085573, "loss": -0.0737, "grad_norm": 8.374490737915039, "learning_rate": 9.033333333333334e-06, "num_tokens": 686384.0, "completions/mean_length": 65.75, "completions/min_length": 45.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.7547735571861267, "rewards/meter/std": 0.39120492339134216, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7428368330001831, "rewards/total_composite/std": 0.4117808938026428, "reward": 0.7428368330001831, "reward_std": 0.4117808938026428, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11435320973396301, "sampling/sampling_logp_difference/max": 0.9131574630737305, "sampling/importance_sampling_ratio/min": 0.4012553095817566, "sampling/importance_sampling_ratio/mean": 1.0321924686431885, "sampling/importance_sampling_ratio/max": 1.9665664434432983, "entropy": 1.4357076957821846, "clip_ratio/low_mean": 0.04342320282012224, "clip_ratio/low_min": 0.04342320282012224, "clip_ratio/high_mean": 0.09262609737925231, "clip_ratio/high_max": 0.09262609737925231, "clip_ratio/region_mean": 0.13604930019937456, "reward_total_mean": 0.7428368330001831, "reward_meter_mean": 0.7547735571861267, "reward_meter_std": 0.39120492339134216, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7428368330001831, "reward_total_composite_std": 0.4117808938026428, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 320.0} {"timestamp_utc": "2026-04-11T20:01:25Z", "mode": "train", "global_step": 321, "epoch": 0.012395736793327155, "loss": 0.0131, "grad_norm": 5.670385837554932, "learning_rate": 9.030303030303031e-06, "num_tokens": 688887.0, "completions/mean_length": 146.875, "completions/min_length": 140.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.875, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.8293708562850952, "rewards/meter/std": 0.18216241896152496, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8293708562850952, "rewards/total_composite/std": 0.18216241896152496, "reward": 0.8293708562850952, "reward_std": 0.18216243386268616, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11904244869947433, "sampling/sampling_logp_difference/max": 1.7790718078613281, "sampling/importance_sampling_ratio/min": 0.16879475116729736, "sampling/importance_sampling_ratio/mean": 1.028894305229187, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4031646698713303, "clip_ratio/low_mean": 0.029992205556482077, "clip_ratio/low_min": 0.029992205556482077, "clip_ratio/high_mean": 0.05931214243173599, "clip_ratio/high_max": 0.05931214243173599, "clip_ratio/region_mean": 0.08930434798821807, "reward_total_mean": 0.8293708562850952, "reward_meter_mean": 0.8293708562850952, "reward_meter_std": 0.18216241896152496, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8293708562850952, "reward_total_composite_std": 0.18216241896152496, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 321.0} {"timestamp_utc": "2026-04-11T20:01:34Z", "mode": "train", "global_step": 322, "epoch": 0.012434352795798579, "loss": -0.1778, "grad_norm": 1.3999040126800537, "learning_rate": 9.027272727272728e-06, "num_tokens": 690651.0, "completions/mean_length": 127.5, "completions/min_length": 66.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 72.5714340209961, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8722488880157471, "rewards/meter/std": 0.34294039011001587, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8692903518676758, "rewards/total_composite/std": 0.3513069450855255, "reward": 0.8692903518676758, "reward_std": 0.3513069152832031, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1413571685552597, "sampling/sampling_logp_difference/max": 1.10498046875, "sampling/importance_sampling_ratio/min": 0.33121734857559204, "sampling/importance_sampling_ratio/mean": 1.0229562520980835, "sampling/importance_sampling_ratio/max": 1.8177605867385864, "entropy": 1.4858018308877945, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10488645080476999, "clip_ratio/high_max": 0.10488645080476999, "clip_ratio/region_mean": 0.10488645080476999, "reward_total_mean": 0.8692903518676758, "reward_meter_mean": 0.8722488880157471, "reward_meter_std": 0.34294039011001587, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8692903518676758, "reward_total_composite_std": 0.3513069450855255, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 322.0} {"timestamp_utc": "2026-04-11T20:01:39Z", "mode": "train", "global_step": 323, "epoch": 0.012472968798270003, "loss": -0.0546, "grad_norm": 8.688957214355469, "learning_rate": 9.024242424242426e-06, "num_tokens": 692669.0, "completions/mean_length": 71.25, "completions/min_length": 58.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9870597720146179, "rewards/meter/std": 0.012377563863992691, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9870597720146179, "rewards/total_composite/std": 0.012377563863992691, "reward": 0.9870597720146179, "reward_std": 0.012377569451928139, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11397970467805862, "sampling/sampling_logp_difference/max": 0.9544820785522461, "sampling/importance_sampling_ratio/min": 0.3850115239620209, "sampling/importance_sampling_ratio/mean": 1.0185412168502808, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1599989607930183, "clip_ratio/low_mean": 0.0364344846457243, "clip_ratio/low_min": 0.0364344846457243, "clip_ratio/high_mean": 0.05869516870006919, "clip_ratio/high_max": 0.05869516870006919, "clip_ratio/region_mean": 0.09512965334579349, "reward_total_mean": 0.9870597720146179, "reward_meter_mean": 0.9870597720146179, "reward_meter_std": 0.012377563863992691, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9870597720146179, "reward_total_composite_std": 0.012377563863992691, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 323.0} {"timestamp_utc": "2026-04-11T20:01:49Z", "mode": "train", "global_step": 324, "epoch": 0.012511584800741427, "loss": -0.1171, "grad_norm": 0.7056026458740234, "learning_rate": 9.021212121212121e-06, "num_tokens": 694598.0, "completions/mean_length": 509.125, "completions/min_length": 489.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.875, "completions/mean_terminated_length": 489.0, "completions/min_terminated_length": 489.0, "completions/max_terminated_length": 489.0, "rewards/meter/mean": 0.5128244161605835, "rewards/meter/std": 0.40451905131340027, "rewards/count_adherence/mean": 0.6180555820465088, "rewards/count_adherence/std": 0.20452874898910522, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/total_composite/mean": 0.33366143703460693, "rewards/total_composite/std": 0.3783239722251892, "reward": 0.33366143703460693, "reward_std": 0.3783239722251892, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07055795192718506, "sampling/sampling_logp_difference/max": 0.9117083549499512, "sampling/importance_sampling_ratio/min": 0.40183717012405396, "sampling/importance_sampling_ratio/mean": 1.012312650680542, "sampling/importance_sampling_ratio/max": 1.636110782623291, "entropy": 0.09911523014307022, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0074130878783762455, "clip_ratio/high_max": 0.0074130878783762455, "clip_ratio/region_mean": 0.0074130878783762455, "reward_total_mean": 0.33366143703460693, "reward_meter_mean": 0.5128244161605835, "reward_meter_std": 0.40451905131340027, "reward_count_adherence_mean": 0.6180555820465088, "reward_count_adherence_std": 0.20452874898910522, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_total_composite_mean": 0.33366143703460693, "reward_total_composite_std": 0.3783239722251892, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 324.0} {"timestamp_utc": "2026-04-11T20:01:55Z", "mode": "train", "global_step": 325, "epoch": 0.012550200803212851, "loss": -0.0167, "grad_norm": 5.355256080627441, "learning_rate": 9.01818181818182e-06, "num_tokens": 697565.0, "completions/mean_length": 173.875, "completions/min_length": 145.0, "completions/max_length": 192.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 173.875, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 192.0, "rewards/meter/mean": 0.7737569212913513, "rewards/meter/std": 0.38154885172843933, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7156921625137329, "rewards/total_composite/std": 0.37208291888237, "reward": 0.7156921625137329, "reward_std": 0.37208291888237, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13555942475795746, "sampling/sampling_logp_difference/max": 1.908766269683838, "sampling/importance_sampling_ratio/min": 0.14826318621635437, "sampling/importance_sampling_ratio/mean": 1.0292820930480957, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7032774537801743, "clip_ratio/low_mean": 0.027904992923140526, "clip_ratio/low_min": 0.027904992923140526, "clip_ratio/high_mean": 0.07463606260716915, "clip_ratio/high_max": 0.07463606260716915, "clip_ratio/region_mean": 0.10254105553030968, "reward_total_mean": 0.7156921625137329, "reward_meter_mean": 0.7737569212913513, "reward_meter_std": 0.38154885172843933, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7156921625137329, "reward_total_composite_std": 0.37208291888237, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 325.0} {"timestamp_utc": "2026-04-11T20:02:05Z", "mode": "train", "global_step": 326, "epoch": 0.012588816805684275, "loss": -0.3057, "grad_norm": 0.9641727209091187, "learning_rate": 9.015151515151516e-06, "num_tokens": 700832.0, "completions/mean_length": 340.375, "completions/min_length": 250.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 283.16668701171875, "completions/min_terminated_length": 250.0, "completions/max_terminated_length": 361.0, "rewards/meter/mean": 0.9441099166870117, "rewards/meter/std": 0.12238472700119019, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.2531938850879669, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.642575740814209, "rewards/total_composite/std": 0.2958056926727295, "reward": 0.642575740814209, "reward_std": 0.2958056628704071, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04643620178103447, "sampling/sampling_logp_difference/max": 1.0232006311416626, "sampling/importance_sampling_ratio/min": 0.35944268107414246, "sampling/importance_sampling_ratio/mean": 1.008855938911438, "sampling/importance_sampling_ratio/max": 1.7436391115188599, "entropy": 0.34993776120245457, "clip_ratio/low_mean": 0.00886194035410881, "clip_ratio/low_min": 0.00886194035410881, "clip_ratio/high_mean": 0.019201413029804826, "clip_ratio/high_max": 0.019201413029804826, "clip_ratio/region_mean": 0.028063353383913636, "reward_total_mean": 0.642575740814209, "reward_meter_mean": 0.9441099166870117, "reward_meter_std": 0.12238472700119019, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.2531938850879669, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.642575740814209, "reward_total_composite_std": 0.2958056926727295, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 326.0} {"timestamp_utc": "2026-04-11T20:02:14Z", "mode": "train", "global_step": 327, "epoch": 0.0126274328081557, "loss": -0.2768, "grad_norm": 0.8649334907531738, "learning_rate": 9.012121212121213e-06, "num_tokens": 704333.0, "completions/mean_length": 309.625, "completions/min_length": 256.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 280.71429443359375, "completions/min_terminated_length": 256.0, "completions/max_terminated_length": 311.0, "rewards/meter/mean": 0.9838745594024658, "rewards/meter/std": 0.026461318135261536, "rewards/count_adherence/mean": 0.7361111640930176, "rewards/count_adherence/std": 0.25845491886138916, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7233796119689941, "rewards/total_composite/std": 0.2528642416000366, "reward": 0.7233796119689941, "reward_std": 0.252864271402359, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05245785042643547, "sampling/sampling_logp_difference/max": 1.7547733783721924, "sampling/importance_sampling_ratio/min": 0.17294643819332123, "sampling/importance_sampling_ratio/mean": 1.010648488998413, "sampling/importance_sampling_ratio/max": 1.7859697341918945, "entropy": 0.48327015712857246, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.03814008738845587, "clip_ratio/high_max": 0.03814008738845587, "clip_ratio/region_mean": 0.03814008738845587, "reward_total_mean": 0.7233796119689941, "reward_meter_mean": 0.9838745594024658, "reward_meter_std": 0.026461318135261536, "reward_count_adherence_mean": 0.7361111640930176, "reward_count_adherence_std": 0.25845491886138916, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7233796119689941, "reward_total_composite_std": 0.2528642416000366, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 327.0} {"timestamp_utc": "2026-04-11T20:02:19Z", "mode": "train", "global_step": 328, "epoch": 0.012666048810627124, "loss": 0.0311, "grad_norm": 11.575224876403809, "learning_rate": 9.00909090909091e-06, "num_tokens": 705878.0, "completions/mean_length": 34.125, "completions/min_length": 31.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.7254409790039062, "rewards/meter/std": 0.31982889771461487, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7254409790039062, "rewards/total_composite/std": 0.31982889771461487, "reward": 0.7254409790039062, "reward_std": 0.31982889771461487, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16250382363796234, "sampling/sampling_logp_difference/max": 1.6080713272094727, "sampling/importance_sampling_ratio/min": 0.2002735137939453, "sampling/importance_sampling_ratio/mean": 1.0245475769042969, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6025570929050446, "clip_ratio/low_mean": 0.05930484738200903, "clip_ratio/low_min": 0.05930484738200903, "clip_ratio/high_mean": 0.11218475922942162, "clip_ratio/high_max": 0.11218475922942162, "clip_ratio/region_mean": 0.17148960661143064, "reward_total_mean": 0.7254409790039062, "reward_meter_mean": 0.7254409790039062, "reward_meter_std": 0.31982889771461487, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7254409790039062, "reward_total_composite_std": 0.31982889771461487, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 328.0} {"timestamp_utc": "2026-04-11T20:02:28Z", "mode": "train", "global_step": 329, "epoch": 0.012704664813098548, "loss": 0.0049, "grad_norm": 4.62138032913208, "learning_rate": 9.006060606060607e-06, "num_tokens": 708351.0, "completions/mean_length": 193.125, "completions/min_length": 136.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 147.57144165039062, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.7495023012161255, "rewards/meter/std": 0.3558323085308075, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7189717292785645, "rewards/total_composite/std": 0.3438011407852173, "reward": 0.7189717292785645, "reward_std": 0.3438011407852173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12337387353181839, "sampling/sampling_logp_difference/max": 1.1887009143829346, "sampling/importance_sampling_ratio/min": 0.3046167492866516, "sampling/importance_sampling_ratio/mean": 1.0160291194915771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2332377284765244, "clip_ratio/low_mean": 0.024105392396450043, "clip_ratio/low_min": 0.024105392396450043, "clip_ratio/high_mean": 0.07188303023576736, "clip_ratio/high_max": 0.07188303023576736, "clip_ratio/region_mean": 0.09598842263221741, "reward_total_mean": 0.7189717292785645, "reward_meter_mean": 0.7495023012161255, "reward_meter_std": 0.3558323085308075, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7189717292785645, "reward_total_composite_std": 0.3438011407852173, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 329.0} {"timestamp_utc": "2026-04-11T20:02:38Z", "mode": "train", "global_step": 330, "epoch": 0.012743280815569972, "loss": -0.155, "grad_norm": 2.0594024658203125, "learning_rate": 9.003030303030303e-06, "num_tokens": 709897.0, "completions/mean_length": 113.25, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 56.28571701049805, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8331777453422546, "rewards/meter/std": 0.3333028256893158, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8302444815635681, "rewards/total_composite/std": 0.34145042300224304, "reward": 0.8302444815635681, "reward_std": 0.34145045280456543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1602991372346878, "sampling/sampling_logp_difference/max": 2.5912528038024902, "sampling/importance_sampling_ratio/min": 0.07492611557245255, "sampling/importance_sampling_ratio/mean": 1.0383025407791138, "sampling/importance_sampling_ratio/max": 1.7736172676086426, "entropy": 1.7517335712909698, "clip_ratio/low_mean": 0.025510204955935478, "clip_ratio/low_min": 0.025510204955935478, "clip_ratio/high_mean": 0.10720423050224781, "clip_ratio/high_max": 0.10720423050224781, "clip_ratio/region_mean": 0.1327144354581833, "reward_total_mean": 0.8302444815635681, "reward_meter_mean": 0.8331777453422546, "reward_meter_std": 0.3333028256893158, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8302444815635681, "reward_total_composite_std": 0.34145042300224304, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 330.0} {"timestamp_utc": "2026-04-11T20:02:43Z", "mode": "train", "global_step": 331, "epoch": 0.012781896818041396, "loss": 0.1271, "grad_norm": 19.114912033081055, "learning_rate": 9e-06, "num_tokens": 712092.0, "completions/mean_length": 72.375, "completions/min_length": 62.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.375, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.7024174928665161, "rewards/meter/std": 0.4215409457683563, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.665668785572052, "rewards/total_composite/std": 0.41643577814102173, "reward": 0.665668785572052, "reward_std": 0.41643577814102173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12084802985191345, "sampling/sampling_logp_difference/max": 1.8433079719543457, "sampling/importance_sampling_ratio/min": 0.15829293429851532, "sampling/importance_sampling_ratio/mean": 0.9930925965309143, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6253799833357334, "clip_ratio/low_mean": 0.03799927420914173, "clip_ratio/low_min": 0.03799927420914173, "clip_ratio/high_mean": 0.06826505670323968, "clip_ratio/high_max": 0.06826505670323968, "clip_ratio/region_mean": 0.10626433091238141, "reward_total_mean": 0.665668785572052, "reward_meter_mean": 0.7024174928665161, "reward_meter_std": 0.4215409457683563, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.665668785572052, "reward_total_composite_std": 0.41643577814102173, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 331.0} {"timestamp_utc": "2026-04-11T20:02:47Z", "mode": "train", "global_step": 332, "epoch": 0.01282051282051282, "loss": -0.0274, "grad_norm": 17.004131317138672, "learning_rate": 8.996969696969697e-06, "num_tokens": 713551.0, "completions/mean_length": 40.375, "completions/min_length": 31.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.4543929100036621, "rewards/meter/std": 0.32974186539649963, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4543929100036621, "rewards/total_composite/std": 0.32974186539649963, "reward": 0.4543929100036621, "reward_std": 0.32974183559417725, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1603153944015503, "sampling/sampling_logp_difference/max": 2.839735984802246, "sampling/importance_sampling_ratio/min": 0.05844109505414963, "sampling/importance_sampling_ratio/mean": 1.0065240859985352, "sampling/importance_sampling_ratio/max": 1.9088724851608276, "entropy": 1.166979692876339, "clip_ratio/low_mean": 0.11972531583160162, "clip_ratio/low_min": 0.11972531583160162, "clip_ratio/high_mean": 0.05861414410173893, "clip_ratio/high_max": 0.05861414410173893, "clip_ratio/region_mean": 0.17833945993334055, "reward_total_mean": 0.4543929100036621, "reward_meter_mean": 0.4543929100036621, "reward_meter_std": 0.32974186539649963, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4543929100036621, "reward_total_composite_std": 0.32974186539649963, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 332.0} {"timestamp_utc": "2026-04-11T20:02:52Z", "mode": "train", "global_step": 333, "epoch": 0.012859128822984244, "loss": 0.0326, "grad_norm": 5.926055908203125, "learning_rate": 8.993939393939395e-06, "num_tokens": 715540.0, "completions/mean_length": 87.625, "completions/min_length": 78.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.625, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.8005719184875488, "rewards/meter/std": 0.2811793088912964, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8005719184875488, "rewards/total_composite/std": 0.2811793088912964, "reward": 0.8005719184875488, "reward_std": 0.2811793386936188, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.093291737139225, "sampling/sampling_logp_difference/max": 2.0903244018554688, "sampling/importance_sampling_ratio/min": 0.12364701926708221, "sampling/importance_sampling_ratio/mean": 1.023775577545166, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8614702522754669, "clip_ratio/low_mean": 0.024796965066343546, "clip_ratio/low_min": 0.024796965066343546, "clip_ratio/high_mean": 0.04428324103355408, "clip_ratio/high_max": 0.04428324103355408, "clip_ratio/region_mean": 0.06908020609989762, "reward_total_mean": 0.8005719184875488, "reward_meter_mean": 0.8005719184875488, "reward_meter_std": 0.2811793088912964, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8005719184875488, "reward_total_composite_std": 0.2811793088912964, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 333.0} {"timestamp_utc": "2026-04-11T20:02:57Z", "mode": "train", "global_step": 334, "epoch": 0.012897744825455668, "loss": 0.0292, "grad_norm": 14.146446228027344, "learning_rate": 8.990909090909092e-06, "num_tokens": 717022.0, "completions/mean_length": 38.25, "completions/min_length": 34.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9717813730239868, "rewards/meter/std": 0.05659351125359535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9717813730239868, "rewards/total_composite/std": 0.05659351125359535, "reward": 0.9717813730239868, "reward_std": 0.05659349635243416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1425103396177292, "sampling/sampling_logp_difference/max": 1.089095115661621, "sampling/importance_sampling_ratio/min": 0.33652088046073914, "sampling/importance_sampling_ratio/mean": 1.0397027730941772, "sampling/importance_sampling_ratio/max": 1.9646681547164917, "entropy": 1.5458678156137466, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/high_mean": 0.13734323251992464, "clip_ratio/high_max": 0.13734323251992464, "clip_ratio/region_mean": 0.15296823251992464, "reward_total_mean": 0.9717813730239868, "reward_meter_mean": 0.9717813730239868, "reward_meter_std": 0.05659351125359535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9717813730239868, "reward_total_composite_std": 0.05659351125359535, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 334.0} {"timestamp_utc": "2026-04-11T20:03:01Z", "mode": "train", "global_step": 335, "epoch": 0.012936360827927092, "loss": -0.0594, "grad_norm": 7.787652969360352, "learning_rate": 8.98787878787879e-06, "num_tokens": 718816.0, "completions/mean_length": 63.25, "completions/min_length": 50.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.25, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.8696156740188599, "rewards/meter/std": 0.31520524621009827, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8696156740188599, "rewards/total_composite/std": 0.31520524621009827, "reward": 0.8696156740188599, "reward_std": 0.31520524621009827, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13351094722747803, "sampling/sampling_logp_difference/max": 1.1205968856811523, "sampling/importance_sampling_ratio/min": 0.32608509063720703, "sampling/importance_sampling_ratio/mean": 1.0193424224853516, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2129319533705711, "clip_ratio/low_mean": 0.01715686358511448, "clip_ratio/low_min": 0.01715686358511448, "clip_ratio/high_mean": 0.10228652320802212, "clip_ratio/high_max": 0.10228652320802212, "clip_ratio/region_mean": 0.1194433867931366, "reward_total_mean": 0.8696156740188599, "reward_meter_mean": 0.8696156740188599, "reward_meter_std": 0.31520524621009827, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8696156740188599, "reward_total_composite_std": 0.31520524621009827, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 335.0} {"timestamp_utc": "2026-04-11T20:03:06Z", "mode": "train", "global_step": 336, "epoch": 0.012974976830398516, "loss": 0.0228, "grad_norm": 7.147242069244385, "learning_rate": 8.984848484848485e-06, "num_tokens": 720702.0, "completions/mean_length": 67.75, "completions/min_length": 59.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9919648766517639, "rewards/meter/std": 0.00817751046270132, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9919648766517639, "rewards/total_composite/std": 0.00817751046270132, "reward": 0.9919648766517639, "reward_std": 0.008177503012120724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10161230713129044, "sampling/sampling_logp_difference/max": 2.4680380821228027, "sampling/importance_sampling_ratio/min": 0.08475096523761749, "sampling/importance_sampling_ratio/mean": 1.0083116292953491, "sampling/importance_sampling_ratio/max": 1.912467360496521, "entropy": 0.7481025010347366, "clip_ratio/low_mean": 0.027916074730455875, "clip_ratio/low_min": 0.027916074730455875, "clip_ratio/high_mean": 0.04853037279099226, "clip_ratio/high_max": 0.04853037279099226, "clip_ratio/region_mean": 0.07644644752144814, "reward_total_mean": 0.9919648766517639, "reward_meter_mean": 0.9919648766517639, "reward_meter_std": 0.00817751046270132, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9919648766517639, "reward_total_composite_std": 0.00817751046270132, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 336.0} {"timestamp_utc": "2026-04-11T20:03:12Z", "mode": "train", "global_step": 337, "epoch": 0.01301359283286994, "loss": 0.3588, "grad_norm": 6.292624473571777, "learning_rate": 8.981818181818182e-06, "num_tokens": 722606.0, "completions/mean_length": 85.0, "completions/min_length": 59.0, "completions/max_length": 170.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.9909840822219849, "rewards/meter/std": 0.0034961404744535685, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8667486310005188, "rewards/total_composite/std": 0.35023483633995056, "reward": 0.8667486310005188, "reward_std": 0.35023483633995056, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07967483252286911, "sampling/sampling_logp_difference/max": 1.7057844400405884, "sampling/importance_sampling_ratio/min": 0.18162985146045685, "sampling/importance_sampling_ratio/mean": 1.014748454093933, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6667735129594803, "clip_ratio/low_mean": 0.007352941203862429, "clip_ratio/low_min": 0.007352941203862429, "clip_ratio/high_mean": 0.07585466234013438, "clip_ratio/high_max": 0.07585466234013438, "clip_ratio/region_mean": 0.08320760354399681, "reward_total_mean": 0.8667486310005188, "reward_meter_mean": 0.9909840822219849, "reward_meter_std": 0.0034961404744535685, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8667486310005188, "reward_total_composite_std": 0.35023483633995056, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 337.0} {"timestamp_utc": "2026-04-11T20:03:17Z", "mode": "train", "global_step": 338, "epoch": 0.013052208835341365, "loss": 0.0094, "grad_norm": 6.0476861000061035, "learning_rate": 8.97878787878788e-06, "num_tokens": 724687.0, "completions/mean_length": 94.125, "completions/min_length": 77.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.125, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.7916698455810547, "rewards/meter/std": 0.20429320633411407, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7916698455810547, "rewards/total_composite/std": 0.20429320633411407, "reward": 0.7916698455810547, "reward_std": 0.2042931765317917, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.051787663251161575, "sampling/sampling_logp_difference/max": 1.4401450157165527, "sampling/importance_sampling_ratio/min": 0.2368934005498886, "sampling/importance_sampling_ratio/mean": 1.0018184185028076, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38130839727818966, "clip_ratio/low_mean": 0.027584444032981992, "clip_ratio/low_min": 0.027584444032981992, "clip_ratio/high_mean": 0.02921690931543708, "clip_ratio/high_max": 0.02921690931543708, "clip_ratio/region_mean": 0.05680135334841907, "reward_total_mean": 0.7916698455810547, "reward_meter_mean": 0.7916698455810547, "reward_meter_std": 0.20429320633411407, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7916698455810547, "reward_total_composite_std": 0.20429320633411407, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 338.0} {"timestamp_utc": "2026-04-11T20:03:22Z", "mode": "train", "global_step": 339, "epoch": 0.013090824837812789, "loss": 0.1049, "grad_norm": 32.10745620727539, "learning_rate": 8.975757575757577e-06, "num_tokens": 726091.0, "completions/mean_length": 25.5, "completions/min_length": 20.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.5, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.45373764634132385, "rewards/meter/std": 0.3389832377433777, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.45373764634132385, "rewards/total_composite/std": 0.3389832377433777, "reward": 0.45373764634132385, "reward_std": 0.3389832377433777, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.163490429520607, "sampling/sampling_logp_difference/max": 1.6797986030578613, "sampling/importance_sampling_ratio/min": 0.186411514878273, "sampling/importance_sampling_ratio/mean": 0.9738427400588989, "sampling/importance_sampling_ratio/max": 1.9982775449752808, "entropy": 0.8039275705814362, "clip_ratio/low_mean": 0.0517513370141387, "clip_ratio/low_min": 0.0517513370141387, "clip_ratio/high_mean": 0.06074862740933895, "clip_ratio/high_max": 0.06074862740933895, "clip_ratio/region_mean": 0.11249996442347765, "reward_total_mean": 0.45373764634132385, "reward_meter_mean": 0.45373764634132385, "reward_meter_std": 0.3389832377433777, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.45373764634132385, "reward_total_composite_std": 0.3389832377433777, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 339.0} {"timestamp_utc": "2026-04-11T20:03:31Z", "mode": "train", "global_step": 340, "epoch": 0.013129440840284215, "loss": -0.0956, "grad_norm": 4.418445110321045, "learning_rate": 8.972727272727272e-06, "num_tokens": 728882.0, "completions/mean_length": 206.875, "completions/min_length": 148.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 163.2857208251953, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.4240845739841461, "rewards/meter/std": 0.30786141753196716, "rewards/count_adherence/mean": 0.8392857313156128, "rewards/count_adherence/std": 0.3452087938785553, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.39793068170547485, "rewards/total_composite/std": 0.29285889863967896, "reward": 0.39793068170547485, "reward_std": 0.29285892844200134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10585647076368332, "sampling/sampling_logp_difference/max": 3.0566635131835938, "sampling/importance_sampling_ratio/min": 0.04704439640045166, "sampling/importance_sampling_ratio/mean": 1.0121982097625732, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5428336411714554, "clip_ratio/low_mean": 0.03718182072043419, "clip_ratio/low_min": 0.03718182072043419, "clip_ratio/high_mean": 0.05819632112979889, "clip_ratio/high_max": 0.05819632112979889, "clip_ratio/region_mean": 0.09537814185023308, "reward_total_mean": 0.39793068170547485, "reward_meter_mean": 0.4240845739841461, "reward_meter_std": 0.30786141753196716, "reward_count_adherence_mean": 0.8392857313156128, "reward_count_adherence_std": 0.3452087938785553, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.39793068170547485, "reward_total_composite_std": 0.29285889863967896, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 340.0} {"timestamp_utc": "2026-04-11T20:03:36Z", "mode": "train", "global_step": 341, "epoch": 0.013168056842755639, "loss": -0.1819, "grad_norm": 7.589391708374023, "learning_rate": 8.969696969696971e-06, "num_tokens": 730886.0, "completions/mean_length": 54.5, "completions/min_length": 25.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9767169952392578, "rewards/meter/std": 0.013271527364850044, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9168034195899963, "rewards/total_composite/std": 0.17712123692035675, "reward": 0.9168034195899963, "reward_std": 0.17712122201919556, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12854255735874176, "sampling/sampling_logp_difference/max": 1.1053946018218994, "sampling/importance_sampling_ratio/min": 0.33108022809028625, "sampling/importance_sampling_ratio/mean": 1.022121787071228, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2029671669006348, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11759159062057734, "clip_ratio/high_max": 0.11759159062057734, "clip_ratio/region_mean": 0.11759159062057734, "reward_total_mean": 0.9168034195899963, "reward_meter_mean": 0.9767169952392578, "reward_meter_std": 0.013271527364850044, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9168034195899963, "reward_total_composite_std": 0.17712123692035675, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 341.0} {"timestamp_utc": "2026-04-11T20:03:46Z", "mode": "train", "global_step": 342, "epoch": 0.013206672845227063, "loss": -0.1304, "grad_norm": 4.1543121337890625, "learning_rate": 8.966666666666667e-06, "num_tokens": 733050.0, "completions/mean_length": 146.5, "completions/min_length": 79.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 94.28572082519531, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.7078612446784973, "rewards/meter/std": 0.3426123261451721, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7020714282989502, "rewards/total_composite/std": 0.3555365204811096, "reward": 0.7020714282989502, "reward_std": 0.3555365204811096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15139326453208923, "sampling/sampling_logp_difference/max": 1.5638463497161865, "sampling/importance_sampling_ratio/min": 0.20932936668395996, "sampling/importance_sampling_ratio/mean": 1.020814299583435, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2604146748781204, "clip_ratio/low_mean": 0.021226415410637856, "clip_ratio/low_min": 0.021226415410637856, "clip_ratio/high_mean": 0.08128050155937672, "clip_ratio/high_max": 0.08128050155937672, "clip_ratio/region_mean": 0.10250691697001457, "reward_total_mean": 0.7020714282989502, "reward_meter_mean": 0.7078612446784973, "reward_meter_std": 0.3426123261451721, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7020714282989502, "reward_total_composite_std": 0.3555365204811096, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 342.0} {"timestamp_utc": "2026-04-11T20:03:51Z", "mode": "train", "global_step": 343, "epoch": 0.013245288847698487, "loss": -0.0077, "grad_norm": 6.76770544052124, "learning_rate": 8.963636363636364e-06, "num_tokens": 735640.0, "completions/mean_length": 136.75, "completions/min_length": 115.0, "completions/max_length": 162.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.75, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 162.0, "rewards/meter/mean": 0.866592288017273, "rewards/meter/std": 0.23036333918571472, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7804628610610962, "rewards/total_composite/std": 0.22920989990234375, "reward": 0.7804628610610962, "reward_std": 0.22920989990234375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04342114180326462, "sampling/sampling_logp_difference/max": 2.3889265060424805, "sampling/importance_sampling_ratio/min": 0.0917281061410904, "sampling/importance_sampling_ratio/mean": 1.0099705457687378, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2882226835936308, "clip_ratio/low_mean": 0.007872846443206072, "clip_ratio/low_min": 0.007872846443206072, "clip_ratio/high_mean": 0.02198702801251784, "clip_ratio/high_max": 0.02198702801251784, "clip_ratio/region_mean": 0.02985987445572391, "reward_total_mean": 0.7804628610610962, "reward_meter_mean": 0.866592288017273, "reward_meter_std": 0.23036333918571472, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7804628610610962, "reward_total_composite_std": 0.22920989990234375, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 343.0} {"timestamp_utc": "2026-04-11T20:04:01Z", "mode": "train", "global_step": 344, "epoch": 0.013283904850169911, "loss": -0.1251, "grad_norm": 0.3268384635448456, "learning_rate": 8.960606060606061e-06, "num_tokens": 738392.0, "completions/mean_length": 511.0, "completions/min_length": 504.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.75, "completions/mean_terminated_length": 508.0, "completions/min_terminated_length": 504.0, "completions/max_terminated_length": 512.0, "rewards/meter/mean": 0.9466683864593506, "rewards/meter/std": 0.14325150847434998, "rewards/count_adherence/mean": 0.736842155456543, "rewards/count_adherence/std": 0.06891091912984848, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6975077390670776, "rewards/total_composite/std": 0.12571200728416443, "reward": 0.6975077390670776, "reward_std": 0.12571200728416443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007342774420976639, "sampling/sampling_logp_difference/max": 0.7591544389724731, "sampling/importance_sampling_ratio/min": 0.46806204319000244, "sampling/importance_sampling_ratio/mean": 1.0004547834396362, "sampling/importance_sampling_ratio/max": 1.704252004623413, "entropy": 0.015514392405748367, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.001968626049347222, "clip_ratio/high_max": 0.001968626049347222, "clip_ratio/region_mean": 0.001968626049347222, "reward_total_mean": 0.6975077390670776, "reward_meter_mean": 0.9466683864593506, "reward_meter_std": 0.14325150847434998, "reward_count_adherence_mean": 0.736842155456543, "reward_count_adherence_std": 0.06891091912984848, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6975077390670776, "reward_total_composite_std": 0.12571200728416443, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 344.0} {"timestamp_utc": "2026-04-11T20:04:07Z", "mode": "train", "global_step": 345, "epoch": 0.013322520852641335, "loss": 0.0533, "grad_norm": 7.332951068878174, "learning_rate": 8.957575757575758e-06, "num_tokens": 740877.0, "completions/mean_length": 135.625, "completions/min_length": 124.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.625, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.8432111740112305, "rewards/meter/std": 0.3098846971988678, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8432111740112305, "rewards/total_composite/std": 0.3098846971988678, "reward": 0.8432111740112305, "reward_std": 0.3098846971988678, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09029541909694672, "sampling/sampling_logp_difference/max": 1.6071224212646484, "sampling/importance_sampling_ratio/min": 0.20046362280845642, "sampling/importance_sampling_ratio/mean": 1.0232460498809814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7220766767859459, "clip_ratio/low_mean": 0.02456349227577448, "clip_ratio/low_min": 0.02456349227577448, "clip_ratio/high_mean": 0.049760105554014444, "clip_ratio/high_max": 0.049760105554014444, "clip_ratio/region_mean": 0.07432359782978892, "reward_total_mean": 0.8432111740112305, "reward_meter_mean": 0.8432111740112305, "reward_meter_std": 0.3098846971988678, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8432111740112305, "reward_total_composite_std": 0.3098846971988678, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 345.0} {"timestamp_utc": "2026-04-11T20:04:13Z", "mode": "train", "global_step": 346, "epoch": 0.01336113685511276, "loss": -0.0646, "grad_norm": 6.021731853485107, "learning_rate": 8.954545454545456e-06, "num_tokens": 743391.0, "completions/mean_length": 136.25, "completions/min_length": 107.0, "completions/max_length": 175.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.25, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 175.0, "rewards/meter/mean": 0.8487316966056824, "rewards/meter/std": 0.22170411050319672, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8351510763168335, "rewards/total_composite/std": 0.2519603967666626, "reward": 0.8351510763168335, "reward_std": 0.2519603669643402, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08980220556259155, "sampling/sampling_logp_difference/max": 1.3795714378356934, "sampling/importance_sampling_ratio/min": 0.2516863942146301, "sampling/importance_sampling_ratio/mean": 1.0095629692077637, "sampling/importance_sampling_ratio/max": 1.87782621383667, "entropy": 0.7361192889511585, "clip_ratio/low_mean": 0.025923453271389008, "clip_ratio/low_min": 0.025923453271389008, "clip_ratio/high_mean": 0.060007378458976746, "clip_ratio/high_max": 0.060007378458976746, "clip_ratio/region_mean": 0.08593083173036575, "reward_total_mean": 0.8351510763168335, "reward_meter_mean": 0.8487316966056824, "reward_meter_std": 0.22170411050319672, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8351510763168335, "reward_total_composite_std": 0.2519603967666626, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 346.0} {"timestamp_utc": "2026-04-11T20:04:18Z", "mode": "train", "global_step": 347, "epoch": 0.013399752857584183, "loss": 0.3161, "grad_norm": 10.403669357299805, "learning_rate": 8.951515151515153e-06, "num_tokens": 745255.0, "completions/mean_length": 84.0, "completions/min_length": 63.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.9903609752655029, "rewards/meter/std": 0.0039792293682694435, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8664380311965942, "rewards/total_composite/std": 0.35011622309684753, "reward": 0.8664380311965942, "reward_std": 0.35011619329452515, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09673066437244415, "sampling/sampling_logp_difference/max": 1.1571464538574219, "sampling/importance_sampling_ratio/min": 0.31438201665878296, "sampling/importance_sampling_ratio/mean": 1.0305798053741455, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0877859592437744, "clip_ratio/low_mean": 0.008223684504628181, "clip_ratio/low_min": 0.008223684504628181, "clip_ratio/high_mean": 0.0906870374456048, "clip_ratio/high_max": 0.0906870374456048, "clip_ratio/region_mean": 0.09891072195023298, "reward_total_mean": 0.8664380311965942, "reward_meter_mean": 0.9903609752655029, "reward_meter_std": 0.0039792293682694435, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8664380311965942, "reward_total_composite_std": 0.35011622309684753, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 347.0} {"timestamp_utc": "2026-04-11T20:04:28Z", "mode": "train", "global_step": 348, "epoch": 0.013438368860055607, "loss": -0.2579, "grad_norm": 1.5081286430358887, "learning_rate": 8.94848484848485e-06, "num_tokens": 748739.0, "completions/mean_length": 268.5, "completions/min_length": 220.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 233.71429443359375, "completions/min_terminated_length": 220.0, "completions/max_terminated_length": 261.0, "rewards/meter/mean": 0.8488904237747192, "rewards/meter/std": 0.3285263478755951, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.24147263169288635, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7942942976951599, "rewards/total_composite/std": 0.33547642827033997, "reward": 0.7942942976951599, "reward_std": 0.33547642827033997, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05050771310925484, "sampling/sampling_logp_difference/max": 2.707244634628296, "sampling/importance_sampling_ratio/min": 0.0667203888297081, "sampling/importance_sampling_ratio/mean": 1.0065252780914307, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3383890204131603, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/high_mean": 0.035924003925174475, "clip_ratio/high_max": 0.035924003925174475, "clip_ratio/region_mean": 0.04728764062747359, "reward_total_mean": 0.7942942976951599, "reward_meter_mean": 0.8488904237747192, "reward_meter_std": 0.3285263478755951, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.24147263169288635, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7942942976951599, "reward_total_composite_std": 0.33547642827033997, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 348.0} {"timestamp_utc": "2026-04-11T20:04:34Z", "mode": "train", "global_step": 349, "epoch": 0.013476984862527032, "loss": 0.0873, "grad_norm": 6.383397102355957, "learning_rate": 8.945454545454546e-06, "num_tokens": 751125.0, "completions/mean_length": 125.25, "completions/min_length": 94.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.25, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.974650502204895, "rewards/meter/std": 0.015549466013908386, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.974650502204895, "rewards/total_composite/std": 0.015549466013908386, "reward": 0.974650502204895, "reward_std": 0.01554945856332779, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1007874459028244, "sampling/sampling_logp_difference/max": 1.4388933181762695, "sampling/importance_sampling_ratio/min": 0.2371901124715805, "sampling/importance_sampling_ratio/mean": 1.01600182056427, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.955632783472538, "clip_ratio/low_mean": 0.025058357510715723, "clip_ratio/low_min": 0.025058357510715723, "clip_ratio/high_mean": 0.07814110023900867, "clip_ratio/high_max": 0.07814110023900867, "clip_ratio/region_mean": 0.10319945774972439, "reward_total_mean": 0.974650502204895, "reward_meter_mean": 0.974650502204895, "reward_meter_std": 0.015549466013908386, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.974650502204895, "reward_total_composite_std": 0.015549466013908386, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 349.0} {"timestamp_utc": "2026-04-11T20:04:39Z", "mode": "train", "global_step": 350, "epoch": 0.013515600864998456, "loss": -0.018, "grad_norm": 14.133708953857422, "learning_rate": 8.942424242424243e-06, "num_tokens": 752644.0, "completions/mean_length": 35.875, "completions/min_length": 31.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.875, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.690624475479126, "rewards/meter/std": 0.37756747007369995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.690624475479126, "rewards/total_composite/std": 0.37756747007369995, "reward": 0.690624475479126, "reward_std": 0.37756747007369995, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1355486959218979, "sampling/sampling_logp_difference/max": 1.4004323482513428, "sampling/importance_sampling_ratio/min": 0.24649037420749664, "sampling/importance_sampling_ratio/mean": 1.0118988752365112, "sampling/importance_sampling_ratio/max": 1.7945036888122559, "entropy": 1.2592712044715881, "clip_ratio/low_mean": 0.044384559616446495, "clip_ratio/low_min": 0.044384559616446495, "clip_ratio/high_mean": 0.07839446794241667, "clip_ratio/high_max": 0.07839446794241667, "clip_ratio/region_mean": 0.12277902755886316, "reward_total_mean": 0.690624475479126, "reward_meter_mean": 0.690624475479126, "reward_meter_std": 0.37756747007369995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.690624475479126, "reward_total_composite_std": 0.37756747007369995, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 350.0} {"timestamp_utc": "2026-04-11T20:06:01Z", "mode": "eval", "global_step": 350, "epoch": 0.013515600864998456, "eval_loss": NaN, "eval_runtime": 82.5629, "eval_samples_per_second": 1.26, "eval_steps_per_second": 0.157, "eval_num_tokens": 752644.0, "eval_completions/mean_length": 212.46153846153845, "eval_completions/min_length": 56.0, "eval_completions/max_length": 442.84615384615387, "eval_completions/clipped_ratio": 0.038461538461538464, "eval_completions/mean_terminated_length": 200.26236431415265, "eval_completions/min_terminated_length": 56.0, "eval_completions/max_terminated_length": 400.0, "eval_rewards/meter/mean": 0.7184457114109626, "eval_rewards/meter/std": 0.3159666336499728, "eval_rewards/count_adherence/mean": 0.8791503814550546, "eval_rewards/count_adherence/std": 0.17575800361541602, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/total_composite/mean": 0.6290297783338107, "eval_rewards/total_composite/std": 0.31780216900201946, "eval_reward": 0.6290297783338107, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03541659864668663, "eval_sampling/sampling_logp_difference/max": 1.040850510964027, "eval_sampling/importance_sampling_ratio/min": 0.3609803124116017, "eval_sampling/importance_sampling_ratio/mean": 1.0086628473722017, "eval_sampling/importance_sampling_ratio/max": 1.47766217818627, "eval_entropy": 0.43929139238137466, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6290297783338107, "eval_reward_meter_mean": 0.7184457114109626, "eval_reward_meter_std": 0.3159666336499728, "eval_reward_count_adherence_mean": 0.8791503814550546, "eval_reward_count_adherence_std": 0.17575800361541602, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_total_composite_mean": 0.6290297783338107, "eval_reward_total_composite_std": 0.31780216900201946, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 350.0} {"timestamp_utc": "2026-04-11T20:06:09Z", "mode": "train", "global_step": 351, "epoch": 0.01355421686746988, "loss": 0.0583, "grad_norm": 7.404175281524658, "learning_rate": 8.93939393939394e-06, "num_tokens": 755121.0, "completions/mean_length": 110.625, "completions/min_length": 102.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.625, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.8419463634490967, "rewards/meter/std": 0.2106313854455948, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8215762376785278, "rewards/total_composite/std": 0.23777370154857635, "reward": 0.8215762376785278, "reward_std": 0.23777368664741516, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10544459521770477, "sampling/sampling_logp_difference/max": 1.7712764739990234, "sampling/importance_sampling_ratio/min": 0.17011570930480957, "sampling/importance_sampling_ratio/mean": 1.0185338258743286, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6541618332266808, "clip_ratio/low_mean": 0.042410715483129025, "clip_ratio/low_min": 0.042410715483129025, "clip_ratio/high_mean": 0.06420147884637117, "clip_ratio/high_max": 0.06420147884637117, "clip_ratio/region_mean": 0.1066121943295002, "reward_total_mean": 0.8215762376785278, "reward_meter_mean": 0.8419463634490967, "reward_meter_std": 0.2106313854455948, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8215762376785278, "reward_total_composite_std": 0.23777370154857635, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 351.0} {"timestamp_utc": "2026-04-11T20:06:18Z", "mode": "train", "global_step": 352, "epoch": 0.013592832869941304, "loss": 0.0229, "grad_norm": 4.749355792999268, "learning_rate": 8.936363636363638e-06, "num_tokens": 759273.0, "completions/mean_length": 312.0, "completions/min_length": 266.0, "completions/max_length": 350.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 312.0, "completions/min_terminated_length": 266.0, "completions/max_terminated_length": 350.0, "rewards/meter/mean": 0.6065003275871277, "rewards/meter/std": 0.3428960144519806, "rewards/count_adherence/mean": 0.8977272510528564, "rewards/count_adherence/std": 0.0758657306432724, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.3508853614330292, "rewards/total_composite/std": 0.3484891951084137, "reward": 0.3508853614330292, "reward_std": 0.3484891951084137, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09127286821603775, "sampling/sampling_logp_difference/max": 3.022757053375244, "sampling/importance_sampling_ratio/min": 0.048666857182979584, "sampling/importance_sampling_ratio/mean": 1.0068981647491455, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5112155750393867, "clip_ratio/low_mean": 0.043291399255394936, "clip_ratio/low_min": 0.043291399255394936, "clip_ratio/high_mean": 0.040051594376564026, "clip_ratio/high_max": 0.040051594376564026, "clip_ratio/region_mean": 0.08334299363195896, "reward_total_mean": 0.3508853614330292, "reward_meter_mean": 0.6065003275871277, "reward_meter_std": 0.3428960144519806, "reward_count_adherence_mean": 0.8977272510528564, "reward_count_adherence_std": 0.0758657306432724, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.3508853614330292, "reward_total_composite_std": 0.3484891951084137, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 352.0} {"timestamp_utc": "2026-04-11T20:06:23Z", "mode": "train", "global_step": 353, "epoch": 0.013631448872412728, "loss": 0.0551, "grad_norm": 15.203680992126465, "learning_rate": 8.933333333333333e-06, "num_tokens": 760972.0, "completions/mean_length": 53.375, "completions/min_length": 49.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.375, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.17881934344768524, "rewards/meter/std": 0.21953943371772766, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.17881934344768524, "rewards/total_composite/std": 0.21953943371772766, "reward": 0.17881934344768524, "reward_std": 0.21953943371772766, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16142068803310394, "sampling/sampling_logp_difference/max": 2.7126896381378174, "sampling/importance_sampling_ratio/min": 0.06635808944702148, "sampling/importance_sampling_ratio/mean": 1.006723403930664, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9390326142311096, "clip_ratio/low_mean": 0.10074498318135738, "clip_ratio/low_min": 0.10074498318135738, "clip_ratio/high_mean": 0.051137037575244904, "clip_ratio/high_max": 0.051137037575244904, "clip_ratio/region_mean": 0.1518820207566023, "reward_total_mean": 0.17881934344768524, "reward_meter_mean": 0.17881934344768524, "reward_meter_std": 0.21953943371772766, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.17881934344768524, "reward_total_composite_std": 0.21953943371772766, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 353.0} {"timestamp_utc": "2026-04-11T20:06:30Z", "mode": "train", "global_step": 354, "epoch": 0.013670064874884152, "loss": 0.0918, "grad_norm": 5.69448184967041, "learning_rate": 8.930303030303032e-06, "num_tokens": 764184.0, "completions/mean_length": 217.5, "completions/min_length": 183.0, "completions/max_length": 255.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 217.5, "completions/min_terminated_length": 183.0, "completions/max_terminated_length": 255.0, "rewards/meter/mean": 0.47910094261169434, "rewards/meter/std": 0.35389965772628784, "rewards/count_adherence/mean": 0.9285714626312256, "rewards/count_adherence/std": 0.07636035233736038, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.43276864290237427, "rewards/total_composite/std": 0.35318341851234436, "reward": 0.43276864290237427, "reward_std": 0.353183388710022, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09170925617218018, "sampling/sampling_logp_difference/max": 4.018363952636719, "sampling/importance_sampling_ratio/min": 0.01798236183822155, "sampling/importance_sampling_ratio/mean": 0.9991462230682373, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4405891187489033, "clip_ratio/low_mean": 0.04170922236517072, "clip_ratio/low_min": 0.04170922236517072, "clip_ratio/high_mean": 0.03562006680294871, "clip_ratio/high_max": 0.03562006680294871, "clip_ratio/region_mean": 0.07732928916811943, "reward_total_mean": 0.43276864290237427, "reward_meter_mean": 0.47910094261169434, "reward_meter_std": 0.35389965772628784, "reward_count_adherence_mean": 0.9285714626312256, "reward_count_adherence_std": 0.07636035233736038, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.43276864290237427, "reward_total_composite_std": 0.35318341851234436, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 354.0} {"timestamp_utc": "2026-04-11T20:06:35Z", "mode": "train", "global_step": 355, "epoch": 0.013708680877355576, "loss": 0.1883, "grad_norm": 8.740988731384277, "learning_rate": 8.927272727272728e-06, "num_tokens": 765870.0, "completions/mean_length": 62.75, "completions/min_length": 50.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.8165422677993774, "rewards/meter/std": 0.2722550332546234, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7710193395614624, "rewards/total_composite/std": 0.3160322606563568, "reward": 0.7710193395614624, "reward_std": 0.3160322606563568, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09419586509466171, "sampling/sampling_logp_difference/max": 1.2416706085205078, "sampling/importance_sampling_ratio/min": 0.2889011800289154, "sampling/importance_sampling_ratio/mean": 0.9960544109344482, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6785079278051853, "clip_ratio/low_mean": 0.013718276750296354, "clip_ratio/low_min": 0.013718276750296354, "clip_ratio/high_mean": 0.0662611536681652, "clip_ratio/high_max": 0.0662611536681652, "clip_ratio/region_mean": 0.07997943041846156, "reward_total_mean": 0.7710193395614624, "reward_meter_mean": 0.8165422677993774, "reward_meter_std": 0.2722550332546234, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7710193395614624, "reward_total_composite_std": 0.3160322606563568, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 355.0} {"timestamp_utc": "2026-04-11T20:06:41Z", "mode": "train", "global_step": 356, "epoch": 0.013747296879827, "loss": 0.3162, "grad_norm": 5.667863368988037, "learning_rate": 8.924242424242425e-06, "num_tokens": 767806.0, "completions/mean_length": 91.0, "completions/min_length": 62.0, "completions/max_length": 174.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 174.0, "rewards/meter/mean": 0.7748513221740723, "rewards/meter/std": 0.3971291780471802, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.4432026445865631, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5287183523178101, "rewards/total_composite/std": 0.4211387634277344, "reward": 0.5287183523178101, "reward_std": 0.421138733625412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.057105425745248795, "sampling/sampling_logp_difference/max": 1.5709530115127563, "sampling/importance_sampling_ratio/min": 0.207846999168396, "sampling/importance_sampling_ratio/mean": 1.00663161277771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4440796282142401, "clip_ratio/low_mean": 0.02032768283970654, "clip_ratio/low_min": 0.02032768283970654, "clip_ratio/high_mean": 0.029830908868461847, "clip_ratio/high_max": 0.029830908868461847, "clip_ratio/region_mean": 0.05015859170816839, "reward_total_mean": 0.5287183523178101, "reward_meter_mean": 0.7748513221740723, "reward_meter_std": 0.3971291780471802, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.4432026445865631, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5287183523178101, "reward_total_composite_std": 0.4211387634277344, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 356.0} {"timestamp_utc": "2026-04-11T20:06:46Z", "mode": "train", "global_step": 357, "epoch": 0.013785912882298424, "loss": -0.0252, "grad_norm": 14.779327392578125, "learning_rate": 8.921212121212122e-06, "num_tokens": 769331.0, "completions/mean_length": 29.625, "completions/min_length": 21.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.644232988357544, "rewards/meter/std": 0.4240405559539795, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.644232988357544, "rewards/total_composite/std": 0.4240405559539795, "reward": 0.644232988357544, "reward_std": 0.4240405559539795, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12809233367443085, "sampling/sampling_logp_difference/max": 1.5906829833984375, "sampling/importance_sampling_ratio/min": 0.20378637313842773, "sampling/importance_sampling_ratio/mean": 1.0188149213790894, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4358574822545052, "clip_ratio/low_mean": 0.05875496123917401, "clip_ratio/low_min": 0.05875496123917401, "clip_ratio/high_mean": 0.07131410390138626, "clip_ratio/high_max": 0.07131410390138626, "clip_ratio/region_mean": 0.13006906514056027, "reward_total_mean": 0.644232988357544, "reward_meter_mean": 0.644232988357544, "reward_meter_std": 0.4240405559539795, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.644232988357544, "reward_total_composite_std": 0.4240405559539795, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 357.0} {"timestamp_utc": "2026-04-11T20:06:56Z", "mode": "train", "global_step": 358, "epoch": 0.013824528884769849, "loss": -0.0033, "grad_norm": 2.133929491043091, "learning_rate": 8.91818181818182e-06, "num_tokens": 771649.0, "completions/mean_length": 177.75, "completions/min_length": 95.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 130.0, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 205.0, "rewards/meter/mean": 0.7501189708709717, "rewards/meter/std": 0.4528391361236572, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.34503278136253357, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6629506349563599, "rewards/total_composite/std": 0.43371596932411194, "reward": 0.6629506349563599, "reward_std": 0.43371593952178955, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.053938575088977814, "sampling/sampling_logp_difference/max": 1.7872505187988281, "sampling/importance_sampling_ratio/min": 0.16741985082626343, "sampling/importance_sampling_ratio/mean": 1.0138229131698608, "sampling/importance_sampling_ratio/max": 1.9287148714065552, "entropy": 0.3757124850526452, "clip_ratio/low_mean": 0.0018292682943865657, "clip_ratio/low_min": 0.0018292682943865657, "clip_ratio/high_mean": 0.053119941614568233, "clip_ratio/high_max": 0.053119941614568233, "clip_ratio/region_mean": 0.0549492099089548, "reward_total_mean": 0.6629506349563599, "reward_meter_mean": 0.7501189708709717, "reward_meter_std": 0.4528391361236572, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.34503278136253357, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6629506349563599, "reward_total_composite_std": 0.43371596932411194, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 358.0} {"timestamp_utc": "2026-04-11T20:07:02Z", "mode": "train", "global_step": 359, "epoch": 0.013863144887241273, "loss": -0.0206, "grad_norm": 4.576873779296875, "learning_rate": 8.915151515151515e-06, "num_tokens": 774739.0, "completions/mean_length": 189.25, "completions/min_length": 167.0, "completions/max_length": 207.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 189.25, "completions/min_terminated_length": 167.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.9521228075027466, "rewards/meter/std": 0.1113143265247345, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9521228075027466, "rewards/total_composite/std": 0.1113143265247345, "reward": 0.9521228075027466, "reward_std": 0.1113143190741539, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033114705234766006, "sampling/sampling_logp_difference/max": 1.1694130897521973, "sampling/importance_sampling_ratio/min": 0.31054913997650146, "sampling/importance_sampling_ratio/mean": 1.0031837224960327, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20762322936207056, "clip_ratio/low_mean": 0.006502890028059483, "clip_ratio/low_min": 0.006502890028059483, "clip_ratio/high_mean": 0.025988655164837837, "clip_ratio/high_max": 0.025988655164837837, "clip_ratio/region_mean": 0.03249154519289732, "reward_total_mean": 0.9521228075027466, "reward_meter_mean": 0.9521228075027466, "reward_meter_std": 0.1113143265247345, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9521228075027466, "reward_total_composite_std": 0.1113143265247345, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 359.0} {"timestamp_utc": "2026-04-11T20:07:07Z", "mode": "train", "global_step": 360, "epoch": 0.013901760889712697, "loss": 0.0254, "grad_norm": 8.154752731323242, "learning_rate": 8.912121212121214e-06, "num_tokens": 776550.0, "completions/mean_length": 69.375, "completions/min_length": 49.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.375, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.7717971801757812, "rewards/meter/std": 0.3203040659427643, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7717971801757812, "rewards/total_composite/std": 0.3203040659427643, "reward": 0.7717971801757812, "reward_std": 0.3203040659427643, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09423349797725677, "sampling/sampling_logp_difference/max": 1.2104854583740234, "sampling/importance_sampling_ratio/min": 0.2980525493621826, "sampling/importance_sampling_ratio/mean": 1.0157387256622314, "sampling/importance_sampling_ratio/max": 1.804463267326355, "entropy": 0.9112853184342384, "clip_ratio/low_mean": 0.031017429195344448, "clip_ratio/low_min": 0.031017429195344448, "clip_ratio/high_mean": 0.06353305862285197, "clip_ratio/high_max": 0.06353305862285197, "clip_ratio/region_mean": 0.09455048781819642, "reward_total_mean": 0.7717971801757812, "reward_meter_mean": 0.7717971801757812, "reward_meter_std": 0.3203040659427643, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7717971801757812, "reward_total_composite_std": 0.3203040659427643, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 360.0} {"timestamp_utc": "2026-04-11T20:07:12Z", "mode": "train", "global_step": 361, "epoch": 0.01394037689218412, "loss": 0.0533, "grad_norm": 13.655211448669434, "learning_rate": 8.90909090909091e-06, "num_tokens": 778655.0, "completions/mean_length": 97.125, "completions/min_length": 73.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.125, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9174784421920776, "rewards/meter/std": 0.10890436172485352, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9174784421920776, "rewards/total_composite/std": 0.10890436172485352, "reward": 0.9174784421920776, "reward_std": 0.10890434682369232, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1321590393781662, "sampling/sampling_logp_difference/max": 1.8137863874435425, "sampling/importance_sampling_ratio/min": 0.16303564608097076, "sampling/importance_sampling_ratio/mean": 1.0211211442947388, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8551923632621765, "clip_ratio/low_mean": 0.026876877062022686, "clip_ratio/low_min": 0.026876877062022686, "clip_ratio/high_mean": 0.07991194678470492, "clip_ratio/high_max": 0.07991194678470492, "clip_ratio/region_mean": 0.10678882384672761, "reward_total_mean": 0.9174784421920776, "reward_meter_mean": 0.9174784421920776, "reward_meter_std": 0.10890436172485352, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9174784421920776, "reward_total_composite_std": 0.10890436172485352, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 361.0} {"timestamp_utc": "2026-04-11T20:07:18Z", "mode": "train", "global_step": 362, "epoch": 0.013978992894655545, "loss": 0.0839, "grad_norm": 3.6917600631713867, "learning_rate": 8.906060606060607e-06, "num_tokens": 781154.0, "completions/mean_length": 111.375, "completions/min_length": 92.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.375, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.9842470288276672, "rewards/meter/std": 0.012328671291470528, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9842470288276672, "rewards/total_composite/std": 0.012328671291470528, "reward": 0.9842470288276672, "reward_std": 0.012328672222793102, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029706332832574844, "sampling/sampling_logp_difference/max": 1.4082119464874268, "sampling/importance_sampling_ratio/min": 0.2445802241563797, "sampling/importance_sampling_ratio/mean": 1.0069833993911743, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20824719313532114, "clip_ratio/low_mean": 0.004234739113599062, "clip_ratio/low_min": 0.004234739113599062, "clip_ratio/high_mean": 0.021009880118072033, "clip_ratio/high_max": 0.021009880118072033, "clip_ratio/region_mean": 0.025244619231671095, "reward_total_mean": 0.9842470288276672, "reward_meter_mean": 0.9842470288276672, "reward_meter_std": 0.012328671291470528, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9842470288276672, "reward_total_composite_std": 0.012328671291470528, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 362.0} {"timestamp_utc": "2026-04-11T20:07:23Z", "mode": "train", "global_step": 363, "epoch": 0.014017608897126969, "loss": 0.0367, "grad_norm": 10.716965675354004, "learning_rate": 8.903030303030304e-06, "num_tokens": 783007.0, "completions/mean_length": 70.625, "completions/min_length": 63.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.6237794160842896, "rewards/meter/std": 0.3550654649734497, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6237794160842896, "rewards/total_composite/std": 0.3550654649734497, "reward": 0.6237794160842896, "reward_std": 0.3550654351711273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10527131706476212, "sampling/sampling_logp_difference/max": 1.7661733627319336, "sampling/importance_sampling_ratio/min": 0.17098604142665863, "sampling/importance_sampling_ratio/mean": 1.0215344429016113, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7273201942443848, "clip_ratio/low_mean": 0.030319053679704666, "clip_ratio/low_min": 0.030319053679704666, "clip_ratio/high_mean": 0.07039879262447357, "clip_ratio/high_max": 0.07039879262447357, "clip_ratio/region_mean": 0.10071784630417824, "reward_total_mean": 0.6237794160842896, "reward_meter_mean": 0.6237794160842896, "reward_meter_std": 0.3550654649734497, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6237794160842896, "reward_total_composite_std": 0.3550654649734497, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 363.0} {"timestamp_utc": "2026-04-11T20:07:28Z", "mode": "train", "global_step": 364, "epoch": 0.014056224899598393, "loss": 0.0098, "grad_norm": 10.589325904846191, "learning_rate": 8.900000000000001e-06, "num_tokens": 784698.0, "completions/mean_length": 55.375, "completions/min_length": 45.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.375, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8658082485198975, "rewards/meter/std": 0.2791600823402405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8658082485198975, "rewards/total_composite/std": 0.2791600823402405, "reward": 0.8658082485198975, "reward_std": 0.2791600525379181, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13043774664402008, "sampling/sampling_logp_difference/max": 1.1420555114746094, "sampling/importance_sampling_ratio/min": 0.3191623091697693, "sampling/importance_sampling_ratio/mean": 1.013712763786316, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1656930446624756, "clip_ratio/low_mean": 0.008928571827709675, "clip_ratio/low_min": 0.008928571827709675, "clip_ratio/high_mean": 0.12814369797706604, "clip_ratio/high_max": 0.12814369797706604, "clip_ratio/region_mean": 0.13707226980477571, "reward_total_mean": 0.8658082485198975, "reward_meter_mean": 0.8658082485198975, "reward_meter_std": 0.2791600823402405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8658082485198975, "reward_total_composite_std": 0.2791600823402405, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 364.0} {"timestamp_utc": "2026-04-11T20:07:34Z", "mode": "train", "global_step": 365, "epoch": 0.014094840902069817, "loss": 0.1186, "grad_norm": 13.111671447753906, "learning_rate": 8.896969696969697e-06, "num_tokens": 787243.0, "completions/mean_length": 122.125, "completions/min_length": 108.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.125, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.07551360130310059, "rewards/meter/std": 0.0628013163805008, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.07550254464149475, "rewards/total_composite/std": 0.06281644105911255, "reward": 0.07550254464149475, "reward_std": 0.06281644105911255, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13020947575569153, "sampling/sampling_logp_difference/max": 2.1617202758789062, "sampling/importance_sampling_ratio/min": 0.11512690037488937, "sampling/importance_sampling_ratio/mean": 1.0089328289031982, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5795126520097256, "clip_ratio/low_mean": 0.05067971581593156, "clip_ratio/low_min": 0.05067971581593156, "clip_ratio/high_mean": 0.0560882892459631, "clip_ratio/high_max": 0.0560882892459631, "clip_ratio/region_mean": 0.10676800506189466, "reward_total_mean": 0.07550254464149475, "reward_meter_mean": 0.07551360130310059, "reward_meter_std": 0.0628013163805008, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.07550254464149475, "reward_total_composite_std": 0.06281644105911255, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 365.0} {"timestamp_utc": "2026-04-11T20:07:39Z", "mode": "train", "global_step": 366, "epoch": 0.014133456904541241, "loss": 0.0132, "grad_norm": 13.758865356445312, "learning_rate": 8.893939393939394e-06, "num_tokens": 788896.0, "completions/mean_length": 53.625, "completions/min_length": 48.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.625, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.5461307764053345, "rewards/meter/std": 0.36441224813461304, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5461307764053345, "rewards/total_composite/std": 0.36441224813461304, "reward": 0.5461307764053345, "reward_std": 0.36441221833229065, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1779501587152481, "sampling/sampling_logp_difference/max": 1.6166021823883057, "sampling/importance_sampling_ratio/min": 0.19857226312160492, "sampling/importance_sampling_ratio/mean": 1.0263890027999878, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.051785424351692, "clip_ratio/low_mean": 0.07220942713320255, "clip_ratio/low_min": 0.07220942713320255, "clip_ratio/high_mean": 0.08675131388008595, "clip_ratio/high_max": 0.08675131388008595, "clip_ratio/region_mean": 0.1589607410132885, "reward_total_mean": 0.5461307764053345, "reward_meter_mean": 0.5461307764053345, "reward_meter_std": 0.36441224813461304, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5461307764053345, "reward_total_composite_std": 0.36441224813461304, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 366.0} {"timestamp_utc": "2026-04-11T20:07:44Z", "mode": "train", "global_step": 367, "epoch": 0.014172072907012665, "loss": -0.0044, "grad_norm": 9.988097190856934, "learning_rate": 8.890909090909091e-06, "num_tokens": 790811.0, "completions/mean_length": 79.375, "completions/min_length": 69.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.375, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9885889291763306, "rewards/meter/std": 0.0046616969630122185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9885889291763306, "rewards/total_composite/std": 0.0046616969630122185, "reward": 0.9885889291763306, "reward_std": 0.004661702550947666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07036007940769196, "sampling/sampling_logp_difference/max": 1.2221425771713257, "sampling/importance_sampling_ratio/min": 0.2945983111858368, "sampling/importance_sampling_ratio/mean": 1.0116159915924072, "sampling/importance_sampling_ratio/max": 1.7074363231658936, "entropy": 0.6636021360754967, "clip_ratio/low_mean": 0.028543358203023672, "clip_ratio/low_min": 0.028543358203023672, "clip_ratio/high_mean": 0.04118129098787904, "clip_ratio/high_max": 0.04118129098787904, "clip_ratio/region_mean": 0.06972464919090271, "reward_total_mean": 0.9885889291763306, "reward_meter_mean": 0.9885889291763306, "reward_meter_std": 0.0046616969630122185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9885889291763306, "reward_total_composite_std": 0.0046616969630122185, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 367.0} {"timestamp_utc": "2026-04-11T20:07:49Z", "mode": "train", "global_step": 368, "epoch": 0.01421068890948409, "loss": -0.0084, "grad_norm": 5.517111778259277, "learning_rate": 8.887878787878789e-06, "num_tokens": 792532.0, "completions/mean_length": 62.125, "completions/min_length": 56.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8935509920120239, "rewards/meter/std": 0.2326761782169342, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8935509920120239, "rewards/total_composite/std": 0.2326761782169342, "reward": 0.8935509920120239, "reward_std": 0.2326761931180954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06450241804122925, "sampling/sampling_logp_difference/max": 1.2612690925598145, "sampling/importance_sampling_ratio/min": 0.2832942605018616, "sampling/importance_sampling_ratio/mean": 0.9973874688148499, "sampling/importance_sampling_ratio/max": 1.846786618232727, "entropy": 0.4450516998767853, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/high_mean": 0.06731301127001643, "clip_ratio/high_max": 0.06731301127001643, "clip_ratio/region_mean": 0.07155029941350222, "reward_total_mean": 0.8935509920120239, "reward_meter_mean": 0.8935509920120239, "reward_meter_std": 0.2326761782169342, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8935509920120239, "reward_total_composite_std": 0.2326761782169342, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 368.0} {"timestamp_utc": "2026-04-11T20:07:54Z", "mode": "train", "global_step": 369, "epoch": 0.014249304911955514, "loss": -0.0431, "grad_norm": 4.8590497970581055, "learning_rate": 8.884848484848486e-06, "num_tokens": 794480.0, "completions/mean_length": 73.5, "completions/min_length": 64.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9953228235244751, "rewards/meter/std": 0.008204305544495583, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9953228235244751, "rewards/total_composite/std": 0.008204305544495583, "reward": 0.9953228235244751, "reward_std": 0.008204314857721329, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.057661134749650955, "sampling/sampling_logp_difference/max": 2.3124313354492188, "sampling/importance_sampling_ratio/min": 0.23970215022563934, "sampling/importance_sampling_ratio/mean": 1.0053033828735352, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4530269633978605, "clip_ratio/low_mean": 0.005859375, "clip_ratio/low_min": 0.005859375, "clip_ratio/high_mean": 0.051239289343357086, "clip_ratio/high_max": 0.051239289343357086, "clip_ratio/region_mean": 0.057098664343357086, "reward_total_mean": 0.9953228235244751, "reward_meter_mean": 0.9953228235244751, "reward_meter_std": 0.008204305544495583, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9953228235244751, "reward_total_composite_std": 0.008204305544495583, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 369.0} {"timestamp_utc": "2026-04-11T20:08:01Z", "mode": "train", "global_step": 370, "epoch": 0.014287920914426938, "loss": -0.0679, "grad_norm": 3.5815742015838623, "learning_rate": 8.881818181818183e-06, "num_tokens": 797320.0, "completions/mean_length": 158.0, "completions/min_length": 135.0, "completions/max_length": 193.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.0, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 193.0, "rewards/meter/mean": 0.9859833717346191, "rewards/meter/std": 0.00827515497803688, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.07715168595314026, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8628234267234802, "rewards/total_composite/std": 0.07765333354473114, "reward": 0.8628234267234802, "reward_std": 0.07765334099531174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036213189363479614, "sampling/sampling_logp_difference/max": 1.2359733581542969, "sampling/importance_sampling_ratio/min": 0.2905518114566803, "sampling/importance_sampling_ratio/mean": 1.002156376838684, "sampling/importance_sampling_ratio/max": 1.8284692764282227, "entropy": 0.33616957906633615, "clip_ratio/low_mean": 0.023962138686329126, "clip_ratio/low_min": 0.023962138686329126, "clip_ratio/high_mean": 0.007220303174108267, "clip_ratio/high_max": 0.007220303174108267, "clip_ratio/region_mean": 0.031182441860437393, "reward_total_mean": 0.8628234267234802, "reward_meter_mean": 0.9859833717346191, "reward_meter_std": 0.00827515497803688, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.07715168595314026, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8628234267234802, "reward_total_composite_std": 0.07765333354473114, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 370.0} {"timestamp_utc": "2026-04-11T20:08:06Z", "mode": "train", "global_step": 371, "epoch": 0.014326536916898362, "loss": -0.0611, "grad_norm": 4.787428379058838, "learning_rate": 8.87878787878788e-06, "num_tokens": 799041.0, "completions/mean_length": 63.125, "completions/min_length": 52.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9925068020820618, "rewards/meter/std": 0.004724609199911356, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9925068020820618, "rewards/total_composite/std": 0.004724609199911356, "reward": 0.9925068020820618, "reward_std": 0.004724607802927494, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03738410025835037, "sampling/sampling_logp_difference/max": 1.3125829696655273, "sampling/importance_sampling_ratio/min": 0.4250510334968567, "sampling/importance_sampling_ratio/mean": 1.0087881088256836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3037525750696659, "clip_ratio/low_mean": 0.02037237980403006, "clip_ratio/low_min": 0.02037237980403006, "clip_ratio/high_mean": 0.02175675635226071, "clip_ratio/high_max": 0.02175675635226071, "clip_ratio/region_mean": 0.04212913615629077, "reward_total_mean": 0.9925068020820618, "reward_meter_mean": 0.9925068020820618, "reward_meter_std": 0.004724609199911356, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9925068020820618, "reward_total_composite_std": 0.004724609199911356, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 371.0} {"timestamp_utc": "2026-04-11T20:08:10Z", "mode": "train", "global_step": 372, "epoch": 0.014365152919369786, "loss": 0.0082, "grad_norm": 10.848628044128418, "learning_rate": 8.875757575757576e-06, "num_tokens": 800618.0, "completions/mean_length": 47.125, "completions/min_length": 44.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7270175218582153, "rewards/meter/std": 0.33816584944725037, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7270175218582153, "rewards/total_composite/std": 0.33816584944725037, "reward": 0.7270175218582153, "reward_std": 0.33816584944725037, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11081711202859879, "sampling/sampling_logp_difference/max": 1.5026662349700928, "sampling/importance_sampling_ratio/min": 0.22253604233264923, "sampling/importance_sampling_ratio/mean": 1.007200002670288, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.453409057110548, "clip_ratio/low_mean": 0.03239734377712011, "clip_ratio/low_min": 0.03239734377712011, "clip_ratio/high_mean": 0.05502570327371359, "clip_ratio/high_max": 0.05502570327371359, "clip_ratio/region_mean": 0.0874230470508337, "reward_total_mean": 0.7270175218582153, "reward_meter_mean": 0.7270175218582153, "reward_meter_std": 0.33816584944725037, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7270175218582153, "reward_total_composite_std": 0.33816584944725037, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 372.0} {"timestamp_utc": "2026-04-11T20:08:15Z", "mode": "train", "global_step": 373, "epoch": 0.014403768921841212, "loss": -0.0215, "grad_norm": 15.01648235321045, "learning_rate": 8.872727272727275e-06, "num_tokens": 802191.0, "completions/mean_length": 51.625, "completions/min_length": 41.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8347938060760498, "rewards/meter/std": 0.2780701220035553, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8347938060760498, "rewards/total_composite/std": 0.2780701220035553, "reward": 0.8347938060760498, "reward_std": 0.2780701220035553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15455831587314606, "sampling/sampling_logp_difference/max": 1.3541040420532227, "sampling/importance_sampling_ratio/min": 0.2581785023212433, "sampling/importance_sampling_ratio/mean": 1.0421912670135498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7377965189516544, "clip_ratio/low_mean": 0.03359880484640598, "clip_ratio/low_min": 0.03359880484640598, "clip_ratio/high_mean": 0.08649335522204638, "clip_ratio/high_max": 0.08649335522204638, "clip_ratio/region_mean": 0.12009216006845236, "reward_total_mean": 0.8347938060760498, "reward_meter_mean": 0.8347938060760498, "reward_meter_std": 0.2780701220035553, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8347938060760498, "reward_total_composite_std": 0.2780701220035553, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 373.0} {"timestamp_utc": "2026-04-11T20:08:20Z", "mode": "train", "global_step": 374, "epoch": 0.014442384924312636, "loss": 0.0391, "grad_norm": 9.044235229492188, "learning_rate": 8.86969696969697e-06, "num_tokens": 804026.0, "completions/mean_length": 77.375, "completions/min_length": 52.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.375, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.8928734064102173, "rewards/meter/std": 0.21326282620429993, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8928734064102173, "rewards/total_composite/std": 0.21326282620429993, "reward": 0.8928734064102173, "reward_std": 0.21326281130313873, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10972950607538223, "sampling/sampling_logp_difference/max": 1.5658352375030518, "sampling/importance_sampling_ratio/min": 0.20891344547271729, "sampling/importance_sampling_ratio/mean": 1.0097415447235107, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7330310307443142, "clip_ratio/low_mean": 0.026901003904640675, "clip_ratio/low_min": 0.026901003904640675, "clip_ratio/high_mean": 0.10351844411343336, "clip_ratio/high_max": 0.10351844411343336, "clip_ratio/region_mean": 0.13041944801807404, "reward_total_mean": 0.8928734064102173, "reward_meter_mean": 0.8928734064102173, "reward_meter_std": 0.21326282620429993, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8928734064102173, "reward_total_composite_std": 0.21326282620429993, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 374.0} {"timestamp_utc": "2026-04-11T20:08:25Z", "mode": "train", "global_step": 375, "epoch": 0.01448100092678406, "loss": 0.0183, "grad_norm": 9.385233879089355, "learning_rate": 8.866666666666668e-06, "num_tokens": 805811.0, "completions/mean_length": 67.125, "completions/min_length": 63.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9862740635871887, "rewards/meter/std": 0.022809647023677826, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9862740635871887, "rewards/total_composite/std": 0.022809647023677826, "reward": 0.9862740635871887, "reward_std": 0.02280966006219387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08005134761333466, "sampling/sampling_logp_difference/max": 1.394679069519043, "sampling/importance_sampling_ratio/min": 0.24791260063648224, "sampling/importance_sampling_ratio/mean": 1.005605697631836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7069630771875381, "clip_ratio/low_mean": 0.009168443502858281, "clip_ratio/low_min": 0.009168443502858281, "clip_ratio/high_mean": 0.051100376760587096, "clip_ratio/high_max": 0.051100376760587096, "clip_ratio/region_mean": 0.06026882026344538, "reward_total_mean": 0.9862740635871887, "reward_meter_mean": 0.9862740635871887, "reward_meter_std": 0.022809647023677826, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9862740635871887, "reward_total_composite_std": 0.022809647023677826, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 375.0} {"timestamp_utc": "2026-04-11T20:08:30Z", "mode": "train", "global_step": 376, "epoch": 0.014519616929255484, "loss": 0.0159, "grad_norm": 7.467469215393066, "learning_rate": 8.863636363636365e-06, "num_tokens": 807817.0, "completions/mean_length": 86.75, "completions/min_length": 77.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.75, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.38385850191116333, "rewards/meter/std": 0.3889022171497345, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.38385850191116333, "rewards/total_composite/std": 0.3889022171497345, "reward": 0.38385850191116333, "reward_std": 0.3889022171497345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14060647785663605, "sampling/sampling_logp_difference/max": 1.3528976440429688, "sampling/importance_sampling_ratio/min": 0.2584901452064514, "sampling/importance_sampling_ratio/mean": 1.0250762701034546, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4634561464190483, "clip_ratio/low_mean": 0.07232486410066485, "clip_ratio/low_min": 0.07232486410066485, "clip_ratio/high_mean": 0.049564928747713566, "clip_ratio/high_max": 0.049564928747713566, "clip_ratio/region_mean": 0.12188979284837842, "reward_total_mean": 0.38385850191116333, "reward_meter_mean": 0.38385850191116333, "reward_meter_std": 0.3889022171497345, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.38385850191116333, "reward_total_composite_std": 0.3889022171497345, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 376.0} {"timestamp_utc": "2026-04-11T20:08:36Z", "mode": "train", "global_step": 377, "epoch": 0.014558232931726908, "loss": -0.0199, "grad_norm": 2.5315823554992676, "learning_rate": 8.860606060606062e-06, "num_tokens": 810779.0, "completions/mean_length": 181.25, "completions/min_length": 162.0, "completions/max_length": 195.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.25, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 195.0, "rewards/meter/mean": 0.8494396209716797, "rewards/meter/std": 0.3451205790042877, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.10681165754795074, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7281081080436707, "rewards/total_composite/std": 0.31404727697372437, "reward": 0.7281081080436707, "reward_std": 0.314047247171402, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04079859331250191, "sampling/sampling_logp_difference/max": 1.434044361114502, "sampling/importance_sampling_ratio/min": 0.2383430302143097, "sampling/importance_sampling_ratio/mean": 1.0015827417373657, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25179606676101685, "clip_ratio/low_mean": 0.004890488460659981, "clip_ratio/low_min": 0.004890488460659981, "clip_ratio/high_mean": 0.028778444975614548, "clip_ratio/high_max": 0.028778444975614548, "clip_ratio/region_mean": 0.03366893343627453, "reward_total_mean": 0.7281081080436707, "reward_meter_mean": 0.8494396209716797, "reward_meter_std": 0.3451205790042877, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.10681165754795074, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7281081080436707, "reward_total_composite_std": 0.31404727697372437, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 377.0} {"timestamp_utc": "2026-04-11T20:08:46Z", "mode": "train", "global_step": 378, "epoch": 0.014596848934198332, "loss": -0.1719, "grad_norm": 6.6282148361206055, "learning_rate": 8.857575757575758e-06, "num_tokens": 812651.0, "completions/mean_length": 143.0, "completions/min_length": 79.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 90.28572082519531, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9918551445007324, "rewards/meter/std": 0.004135291092097759, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.24800792336463928, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8677093982696533, "rewards/total_composite/std": 0.24586959183216095, "reward": 0.8677093982696533, "reward_std": 0.24586959183216095, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.052827268838882446, "sampling/sampling_logp_difference/max": 1.2060942649841309, "sampling/importance_sampling_ratio/min": 0.29936423897743225, "sampling/importance_sampling_ratio/mean": 1.008686900138855, "sampling/importance_sampling_ratio/max": 1.980355143547058, "entropy": 0.33806798979640007, "clip_ratio/low_mean": 0.008928571827709675, "clip_ratio/low_min": 0.008928571827709675, "clip_ratio/high_mean": 0.0347325021866709, "clip_ratio/high_max": 0.0347325021866709, "clip_ratio/region_mean": 0.043661074014380574, "reward_total_mean": 0.8677093982696533, "reward_meter_mean": 0.9918551445007324, "reward_meter_std": 0.004135291092097759, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.24800792336463928, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8677093982696533, "reward_total_composite_std": 0.24586959183216095, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 378.0} {"timestamp_utc": "2026-04-11T20:08:56Z", "mode": "train", "global_step": 379, "epoch": 0.014635464936669756, "loss": -0.1795, "grad_norm": 1.1005628108978271, "learning_rate": 8.854545454545455e-06, "num_tokens": 814504.0, "completions/mean_length": 130.625, "completions/min_length": 65.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 76.14286041259766, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9917647838592529, "rewards/meter/std": 0.009205368347465992, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9303869605064392, "rewards/total_composite/std": 0.17772506177425385, "reward": 0.9303869605064392, "reward_std": 0.17772506177425385, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08945769816637039, "sampling/sampling_logp_difference/max": 1.1567516326904297, "sampling/importance_sampling_ratio/min": 0.3145061433315277, "sampling/importance_sampling_ratio/mean": 1.0020660161972046, "sampling/importance_sampling_ratio/max": 1.736937165260315, "entropy": 0.7669309824705124, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08018230739980936, "clip_ratio/high_max": 0.08018230739980936, "clip_ratio/region_mean": 0.08018230739980936, "reward_total_mean": 0.9303869605064392, "reward_meter_mean": 0.9917647838592529, "reward_meter_std": 0.009205368347465992, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9303869605064392, "reward_total_composite_std": 0.17772506177425385, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 379.0} {"timestamp_utc": "2026-04-11T20:09:01Z", "mode": "train", "global_step": 380, "epoch": 0.01467408093914118, "loss": -0.0121, "grad_norm": 9.77833366394043, "learning_rate": 8.851515151515152e-06, "num_tokens": 816336.0, "completions/mean_length": 62.0, "completions/min_length": 52.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8162009716033936, "rewards/meter/std": 0.21702320873737335, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8162009716033936, "rewards/total_composite/std": 0.21702320873737335, "reward": 0.8162009716033936, "reward_std": 0.21702319383621216, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13395410776138306, "sampling/sampling_logp_difference/max": 1.2279243469238281, "sampling/importance_sampling_ratio/min": 0.2928999066352844, "sampling/importance_sampling_ratio/mean": 1.0254578590393066, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2989665120840073, "clip_ratio/low_mean": 0.05674390681087971, "clip_ratio/low_min": 0.05674390681087971, "clip_ratio/high_mean": 0.07530082948505878, "clip_ratio/high_max": 0.07530082948505878, "clip_ratio/region_mean": 0.1320447362959385, "reward_total_mean": 0.8162009716033936, "reward_meter_mean": 0.8162009716033936, "reward_meter_std": 0.21702320873737335, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8162009716033936, "reward_total_composite_std": 0.21702320873737335, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 380.0} {"timestamp_utc": "2026-04-11T20:09:06Z", "mode": "train", "global_step": 381, "epoch": 0.014712696941612605, "loss": 0.0106, "grad_norm": 5.939858913421631, "learning_rate": 8.84848484848485e-06, "num_tokens": 818123.0, "completions/mean_length": 77.375, "completions/min_length": 69.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.375, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9950414299964905, "rewards/meter/std": 0.005652880761772394, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9950414299964905, "rewards/total_composite/std": 0.005652880761772394, "reward": 0.9950414299964905, "reward_std": 0.005652868654578924, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05509962514042854, "sampling/sampling_logp_difference/max": 1.6830365657806396, "sampling/importance_sampling_ratio/min": 0.18580889701843262, "sampling/importance_sampling_ratio/mean": 1.0126311779022217, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47446247562766075, "clip_ratio/low_mean": 0.02306468691676855, "clip_ratio/low_min": 0.02306468691676855, "clip_ratio/high_mean": 0.02148950519040227, "clip_ratio/high_max": 0.02148950519040227, "clip_ratio/region_mean": 0.04455419210717082, "reward_total_mean": 0.9950414299964905, "reward_meter_mean": 0.9950414299964905, "reward_meter_std": 0.005652880761772394, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9950414299964905, "reward_total_composite_std": 0.005652880761772394, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 381.0} {"timestamp_utc": "2026-04-11T20:09:16Z", "mode": "train", "global_step": 382, "epoch": 0.014751312944084029, "loss": -0.2242, "grad_norm": 1.8641679286956787, "learning_rate": 8.845454545454547e-06, "num_tokens": 820713.0, "completions/mean_length": 211.75, "completions/min_length": 158.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 168.85714721679688, "completions/min_terminated_length": 158.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.9491947889328003, "rewards/meter/std": 0.10156966745853424, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.23299294710159302, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7125877141952515, "rewards/total_composite/std": 0.2403973489999771, "reward": 0.7125877141952515, "reward_std": 0.2403973639011383, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04226767271757126, "sampling/sampling_logp_difference/max": 1.4181485176086426, "sampling/importance_sampling_ratio/min": 0.24216197431087494, "sampling/importance_sampling_ratio/mean": 1.0095936059951782, "sampling/importance_sampling_ratio/max": 1.9695637226104736, "entropy": 0.311492882668972, "clip_ratio/low_mean": 0.0074404762126505375, "clip_ratio/low_min": 0.0074404762126505375, "clip_ratio/high_mean": 0.023234429769217968, "clip_ratio/high_max": 0.023234429769217968, "clip_ratio/region_mean": 0.030674905981868505, "reward_total_mean": 0.7125877141952515, "reward_meter_mean": 0.9491947889328003, "reward_meter_std": 0.10156966745853424, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.23299294710159302, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7125877141952515, "reward_total_composite_std": 0.2403973489999771, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 382.0} {"timestamp_utc": "2026-04-11T20:09:26Z", "mode": "train", "global_step": 383, "epoch": 0.014789928946555453, "loss": -0.1634, "grad_norm": 1.574141263961792, "learning_rate": 8.842424242424244e-06, "num_tokens": 823881.0, "completions/mean_length": 263.0, "completions/min_length": 199.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 227.4285888671875, "completions/min_terminated_length": 199.0, "completions/max_terminated_length": 252.0, "rewards/meter/mean": 0.8664563298225403, "rewards/meter/std": 0.2947894036769867, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5856432914733887, "rewards/total_composite/std": 0.2903149425983429, "reward": 0.5856432914733887, "reward_std": 0.2903149425983429, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01734515093266964, "sampling/sampling_logp_difference/max": 1.1482505798339844, "sampling/importance_sampling_ratio/min": 0.3171911835670471, "sampling/importance_sampling_ratio/mean": 1.000448226928711, "sampling/importance_sampling_ratio/max": 1.8257579803466797, "entropy": 0.10115705709904432, "clip_ratio/low_mean": 0.0015822785208001733, "clip_ratio/low_min": 0.0015822785208001733, "clip_ratio/high_mean": 0.01344697322929278, "clip_ratio/high_max": 0.01344697322929278, "clip_ratio/region_mean": 0.015029251750092953, "reward_total_mean": 0.5856432914733887, "reward_meter_mean": 0.8664563298225403, "reward_meter_std": 0.2947894036769867, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5856432914733887, "reward_total_composite_std": 0.2903149425983429, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 383.0} {"timestamp_utc": "2026-04-11T20:09:36Z", "mode": "train", "global_step": 384, "epoch": 0.014828544949026877, "loss": -0.0853, "grad_norm": 2.7595345973968506, "learning_rate": 8.83939393939394e-06, "num_tokens": 825706.0, "completions/mean_length": 132.125, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 77.85714721679688, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.8857226967811584, "rewards/meter/std": 0.2998782992362976, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8233171105384827, "rewards/total_composite/std": 0.32403311133384705, "reward": 0.8233171105384827, "reward_std": 0.32403311133384705, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05464348942041397, "sampling/sampling_logp_difference/max": 1.6103429794311523, "sampling/importance_sampling_ratio/min": 0.19981907308101654, "sampling/importance_sampling_ratio/mean": 1.005861759185791, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3592350035905838, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/high_mean": 0.03935463528614491, "clip_ratio/high_max": 0.03935463528614491, "clip_ratio/region_mean": 0.04791627905797213, "reward_total_mean": 0.8233171105384827, "reward_meter_mean": 0.8857226967811584, "reward_meter_std": 0.2998782992362976, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8233171105384827, "reward_total_composite_std": 0.32403311133384705, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 384.0} {"timestamp_utc": "2026-04-11T20:09:46Z", "mode": "train", "global_step": 385, "epoch": 0.014867160951498301, "loss": -0.0509, "grad_norm": 2.582751512527466, "learning_rate": 8.836363636363637e-06, "num_tokens": 827248.0, "completions/mean_length": 178.75, "completions/min_length": 60.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 67.66667175292969, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.714938223361969, "rewards/meter/std": 0.36748242378234863, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.714938223361969, "rewards/total_composite/std": 0.36748242378234863, "reward": 0.714938223361969, "reward_std": 0.36748242378234863, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07517319172620773, "sampling/sampling_logp_difference/max": 1.0834226608276367, "sampling/importance_sampling_ratio/min": 0.33843517303466797, "sampling/importance_sampling_ratio/mean": 1.0119549036026, "sampling/importance_sampling_ratio/max": 1.566494107246399, "entropy": 0.43436889722943306, "clip_ratio/low_mean": 0.003689236124046147, "clip_ratio/low_min": 0.003689236124046147, "clip_ratio/high_mean": 0.041496834717690945, "clip_ratio/high_max": 0.041496834717690945, "clip_ratio/region_mean": 0.04518607084173709, "reward_total_mean": 0.714938223361969, "reward_meter_mean": 0.714938223361969, "reward_meter_std": 0.36748242378234863, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.714938223361969, "reward_total_composite_std": 0.36748242378234863, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 385.0} {"timestamp_utc": "2026-04-11T20:09:50Z", "mode": "train", "global_step": 386, "epoch": 0.014905776953969725, "loss": 0.0492, "grad_norm": 13.948408126831055, "learning_rate": 8.833333333333334e-06, "num_tokens": 828954.0, "completions/mean_length": 58.25, "completions/min_length": 42.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.690985918045044, "rewards/meter/std": 0.42123687267303467, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.690985918045044, "rewards/total_composite/std": 0.42123687267303467, "reward": 0.690985918045044, "reward_std": 0.42123687267303467, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12392318993806839, "sampling/sampling_logp_difference/max": 1.698615312576294, "sampling/importance_sampling_ratio/min": 0.1829366534948349, "sampling/importance_sampling_ratio/mean": 1.0141727924346924, "sampling/importance_sampling_ratio/max": 1.8247418403625488, "entropy": 1.1986883580684662, "clip_ratio/low_mean": 0.03804087173193693, "clip_ratio/low_min": 0.03804087173193693, "clip_ratio/high_mean": 0.07191554573364556, "clip_ratio/high_max": 0.07191554573364556, "clip_ratio/region_mean": 0.10995641746558249, "reward_total_mean": 0.690985918045044, "reward_meter_mean": 0.690985918045044, "reward_meter_std": 0.42123687267303467, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.690985918045044, "reward_total_composite_std": 0.42123687267303467, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 386.0} {"timestamp_utc": "2026-04-11T20:10:00Z", "mode": "train", "global_step": 387, "epoch": 0.01494439295644115, "loss": -0.2826, "grad_norm": 0.3322877585887909, "learning_rate": 8.830303030303031e-06, "num_tokens": 832724.0, "completions/mean_length": 315.25, "completions/min_length": 254.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 287.14288330078125, "completions/min_terminated_length": 254.0, "completions/max_terminated_length": 306.0, "rewards/meter/mean": 0.9916845560073853, "rewards/meter/std": 0.01541493646800518, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.2121320217847824, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6194590330123901, "rewards/total_composite/std": 0.21031685173511505, "reward": 0.6194590330123901, "reward_std": 0.21031685173511505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011599292047321796, "sampling/sampling_logp_difference/max": 1.863002061843872, "sampling/importance_sampling_ratio/min": 0.15520599484443665, "sampling/importance_sampling_ratio/mean": 1.0010322332382202, "sampling/importance_sampling_ratio/max": 1.4588537216186523, "entropy": 0.06537747196853161, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007820297265425324, "clip_ratio/high_max": 0.007820297265425324, "clip_ratio/region_mean": 0.007820297265425324, "reward_total_mean": 0.6194590330123901, "reward_meter_mean": 0.9916845560073853, "reward_meter_std": 0.01541493646800518, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.2121320217847824, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6194590330123901, "reward_total_composite_std": 0.21031685173511505, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 387.0} {"timestamp_utc": "2026-04-11T20:10:06Z", "mode": "train", "global_step": 388, "epoch": 0.014983008958912573, "loss": 0.0025, "grad_norm": 3.59706449508667, "learning_rate": 8.827272727272727e-06, "num_tokens": 834907.0, "completions/mean_length": 92.875, "completions/min_length": 86.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.875, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9513072967529297, "rewards/meter/std": 0.04710450395941734, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9105833172798157, "rewards/total_composite/std": 0.11427941173315048, "reward": 0.9105833172798157, "reward_std": 0.11427939683198929, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02844572812318802, "sampling/sampling_logp_difference/max": 1.0414361953735352, "sampling/importance_sampling_ratio/min": 0.3529474139213562, "sampling/importance_sampling_ratio/mean": 1.0046768188476562, "sampling/importance_sampling_ratio/max": 1.8268436193466187, "entropy": 0.24626289308071136, "clip_ratio/low_mean": 0.00827294704504311, "clip_ratio/low_min": 0.00827294704504311, "clip_ratio/high_mean": 0.01489788806065917, "clip_ratio/high_max": 0.01489788806065917, "clip_ratio/region_mean": 0.02317083510570228, "reward_total_mean": 0.9105833172798157, "reward_meter_mean": 0.9513072967529297, "reward_meter_std": 0.04710450395941734, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9105833172798157, "reward_total_composite_std": 0.11427941173315048, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 388.0} {"timestamp_utc": "2026-04-11T20:10:10Z", "mode": "train", "global_step": 389, "epoch": 0.015021624961383997, "loss": 0.0368, "grad_norm": 10.221504211425781, "learning_rate": 8.824242424242426e-06, "num_tokens": 836445.0, "completions/mean_length": 34.25, "completions/min_length": 27.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8630082011222839, "rewards/meter/std": 0.34829336404800415, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8630082011222839, "rewards/total_composite/std": 0.34829336404800415, "reward": 0.8630082011222839, "reward_std": 0.34829336404800415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11211102455854416, "sampling/sampling_logp_difference/max": 2.2958288192749023, "sampling/importance_sampling_ratio/min": 0.10067791491746902, "sampling/importance_sampling_ratio/mean": 1.0091283321380615, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.777605514973402, "clip_ratio/low_mean": 0.006756756920367479, "clip_ratio/low_min": 0.006756756920367479, "clip_ratio/high_mean": 0.09026568429544568, "clip_ratio/high_max": 0.09026568429544568, "clip_ratio/region_mean": 0.09702244121581316, "reward_total_mean": 0.8630082011222839, "reward_meter_mean": 0.8630082011222839, "reward_meter_std": 0.34829336404800415, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8630082011222839, "reward_total_composite_std": 0.34829336404800415, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 389.0} {"timestamp_utc": "2026-04-11T20:10:15Z", "mode": "train", "global_step": 390, "epoch": 0.015060240963855422, "loss": -0.0172, "grad_norm": 8.775293350219727, "learning_rate": 8.821212121212121e-06, "num_tokens": 838331.0, "completions/mean_length": 54.75, "completions/min_length": 44.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.48685652017593384, "rewards/meter/std": 0.3812946379184723, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.48685652017593384, "rewards/total_composite/std": 0.3812946379184723, "reward": 0.48685652017593384, "reward_std": 0.3812946081161499, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12760156393051147, "sampling/sampling_logp_difference/max": 1.3602138757705688, "sampling/importance_sampling_ratio/min": 0.2566058933734894, "sampling/importance_sampling_ratio/mean": 1.029207468032837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3469589613378048, "clip_ratio/low_mean": 0.030689293053001165, "clip_ratio/low_min": 0.030689293053001165, "clip_ratio/high_mean": 0.07300483155995607, "clip_ratio/high_max": 0.07300483155995607, "clip_ratio/region_mean": 0.10369412461295724, "reward_total_mean": 0.48685652017593384, "reward_meter_mean": 0.48685652017593384, "reward_meter_std": 0.3812946379184723, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.48685652017593384, "reward_total_composite_std": 0.3812946379184723, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 390.0} {"timestamp_utc": "2026-04-11T20:10:26Z", "mode": "train", "global_step": 391, "epoch": 0.015098856966326846, "loss": -0.1899, "grad_norm": 0.7848347425460815, "learning_rate": 8.818181818181819e-06, "num_tokens": 842434.0, "completions/mean_length": 342.875, "completions/min_length": 295.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 318.71429443359375, "completions/min_terminated_length": 295.0, "completions/max_terminated_length": 359.0, "rewards/meter/mean": 0.7168420553207397, "rewards/meter/std": 0.3972436785697937, "rewards/count_adherence/mean": 0.6136363744735718, "rewards/count_adherence/std": 0.25132471323013306, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5062733888626099, "rewards/total_composite/std": 0.2856999337673187, "reward": 0.5062733888626099, "reward_std": 0.2856999337673187, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008826361037790775, "sampling/sampling_logp_difference/max": 1.076157569885254, "sampling/importance_sampling_ratio/min": 0.3409028947353363, "sampling/importance_sampling_ratio/mean": 1.0016242265701294, "sampling/importance_sampling_ratio/max": 1.6002336740493774, "entropy": 0.051004831213504076, "clip_ratio/low_mean": 0.003784051747061312, "clip_ratio/low_min": 0.003784051747061312, "clip_ratio/high_mean": 0.0027604734350461513, "clip_ratio/high_max": 0.0027604734350461513, "clip_ratio/region_mean": 0.0065445251821074635, "reward_total_mean": 0.5062733888626099, "reward_meter_mean": 0.7168420553207397, "reward_meter_std": 0.3972436785697937, "reward_count_adherence_mean": 0.6136363744735718, "reward_count_adherence_std": 0.25132471323013306, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5062733888626099, "reward_total_composite_std": 0.2856999337673187, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 391.0} {"timestamp_utc": "2026-04-11T20:10:36Z", "mode": "train", "global_step": 392, "epoch": 0.01513747296879827, "loss": -0.3491, "grad_norm": 0.6587200164794922, "learning_rate": 8.815151515151516e-06, "num_tokens": 845556.0, "completions/mean_length": 373.25, "completions/min_length": 283.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 290.0, "completions/min_terminated_length": 283.0, "completions/max_terminated_length": 304.0, "rewards/meter/mean": 0.8696821331977844, "rewards/meter/std": 0.35143736004829407, "rewards/count_adherence/mean": 0.5277777910232544, "rewards/count_adherence/std": 0.37912923097610474, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5244123935699463, "rewards/total_composite/std": 0.3767135441303253, "reward": 0.5244123935699463, "reward_std": 0.3767135441303253, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015206166543066502, "sampling/sampling_logp_difference/max": 1.0871790647506714, "sampling/importance_sampling_ratio/min": 0.33716630935668945, "sampling/importance_sampling_ratio/mean": 1.002270221710205, "sampling/importance_sampling_ratio/max": 1.8669248819351196, "entropy": 0.05020246421918273, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007834467338398099, "clip_ratio/high_max": 0.007834467338398099, "clip_ratio/region_mean": 0.007834467338398099, "reward_total_mean": 0.5244123935699463, "reward_meter_mean": 0.8696821331977844, "reward_meter_std": 0.35143736004829407, "reward_count_adherence_mean": 0.5277777910232544, "reward_count_adherence_std": 0.37912923097610474, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5244123935699463, "reward_total_composite_std": 0.3767135441303253, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 392.0} {"timestamp_utc": "2026-04-11T20:10:41Z", "mode": "train", "global_step": 393, "epoch": 0.015176088971269694, "loss": 0.0322, "grad_norm": 8.689268112182617, "learning_rate": 8.812121212121213e-06, "num_tokens": 847311.0, "completions/mean_length": 76.375, "completions/min_length": 67.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.375, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.8588340878486633, "rewards/meter/std": 0.2888956665992737, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8588340878486633, "rewards/total_composite/std": 0.2888956665992737, "reward": 0.8588340878486633, "reward_std": 0.2888956665992737, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07139719277620316, "sampling/sampling_logp_difference/max": 1.9691343307495117, "sampling/importance_sampling_ratio/min": 0.13957762718200684, "sampling/importance_sampling_ratio/mean": 0.9985421895980835, "sampling/importance_sampling_ratio/max": 1.9332025051116943, "entropy": 0.4674673527479172, "clip_ratio/low_mean": 0.010939412750303745, "clip_ratio/low_min": 0.010939412750303745, "clip_ratio/high_mean": 0.06486599263735116, "clip_ratio/high_max": 0.06486599263735116, "clip_ratio/region_mean": 0.0758054053876549, "reward_total_mean": 0.8588340878486633, "reward_meter_mean": 0.8588340878486633, "reward_meter_std": 0.2888956665992737, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8588340878486633, "reward_total_composite_std": 0.2888956665992737, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 393.0} {"timestamp_utc": "2026-04-11T20:10:46Z", "mode": "train", "global_step": 394, "epoch": 0.015214704973741118, "loss": 0.0681, "grad_norm": 4.890065670013428, "learning_rate": 8.809090909090909e-06, "num_tokens": 849152.0, "completions/mean_length": 66.125, "completions/min_length": 59.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9758315086364746, "rewards/meter/std": 0.012030374258756638, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9758315086364746, "rewards/total_composite/std": 0.012030374258756638, "reward": 0.9758315086364746, "reward_std": 0.012030377984046936, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.048741716891527176, "sampling/sampling_logp_difference/max": 0.8468852043151855, "sampling/importance_sampling_ratio/min": 0.4598883390426636, "sampling/importance_sampling_ratio/mean": 1.004264235496521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37163497880101204, "clip_ratio/low_mean": 0.007181095774285495, "clip_ratio/low_min": 0.007181095774285495, "clip_ratio/high_mean": 0.028151679784059525, "clip_ratio/high_max": 0.028151679784059525, "clip_ratio/region_mean": 0.03533277555834502, "reward_total_mean": 0.9758315086364746, "reward_meter_mean": 0.9758315086364746, "reward_meter_std": 0.012030374258756638, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9758315086364746, "reward_total_composite_std": 0.012030374258756638, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 394.0} {"timestamp_utc": "2026-04-11T20:10:54Z", "mode": "train", "global_step": 395, "epoch": 0.015253320976212542, "loss": -0.0587, "grad_norm": 2.795933485031128, "learning_rate": 8.806060606060608e-06, "num_tokens": 852613.0, "completions/mean_length": 228.625, "completions/min_length": 186.0, "completions/max_length": 265.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 228.625, "completions/min_terminated_length": 186.0, "completions/max_terminated_length": 265.0, "rewards/meter/mean": 0.9906454086303711, "rewards/meter/std": 0.012306920252740383, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625820279121399, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8257385492324829, "rewards/total_composite/std": 0.3412371277809143, "reward": 0.8257385492324829, "reward_std": 0.3412370979785919, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02395275980234146, "sampling/sampling_logp_difference/max": 1.5854034423828125, "sampling/importance_sampling_ratio/min": 0.3412272036075592, "sampling/importance_sampling_ratio/mean": 1.0015069246292114, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17339316615834832, "clip_ratio/low_mean": 0.009408602491021156, "clip_ratio/low_min": 0.009408602491021156, "clip_ratio/high_mean": 0.012422295869328082, "clip_ratio/high_max": 0.012422295869328082, "clip_ratio/region_mean": 0.021830898360349238, "reward_total_mean": 0.8257385492324829, "reward_meter_mean": 0.9906454086303711, "reward_meter_std": 0.012306920252740383, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625820279121399, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8257385492324829, "reward_total_composite_std": 0.3412371277809143, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 395.0} {"timestamp_utc": "2026-04-11T20:10:59Z", "mode": "train", "global_step": 396, "epoch": 0.015291936978683966, "loss": -0.0516, "grad_norm": 15.071035385131836, "learning_rate": 8.803030303030303e-06, "num_tokens": 854021.0, "completions/mean_length": 35.0, "completions/min_length": 27.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9902995824813843, "rewards/meter/std": 0.008138173259794712, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9902995824813843, "rewards/total_composite/std": 0.008138173259794712, "reward": 0.9902995824813843, "reward_std": 0.008138181641697884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08739858120679855, "sampling/sampling_logp_difference/max": 1.2736406326293945, "sampling/importance_sampling_ratio/min": 0.2798110842704773, "sampling/importance_sampling_ratio/mean": 1.0173804759979248, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7259577997028828, "clip_ratio/low_mean": 0.03667238587513566, "clip_ratio/low_min": 0.03667238587513566, "clip_ratio/high_mean": 0.034834470599889755, "clip_ratio/high_max": 0.034834470599889755, "clip_ratio/region_mean": 0.07150685647502542, "reward_total_mean": 0.9902995824813843, "reward_meter_mean": 0.9902995824813843, "reward_meter_std": 0.008138173259794712, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9902995824813843, "reward_total_composite_std": 0.008138173259794712, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 396.0} {"timestamp_utc": "2026-04-11T20:11:09Z", "mode": "train", "global_step": 397, "epoch": 0.01533055298115539, "loss": 0.0085, "grad_norm": 4.2064313888549805, "learning_rate": 8.8e-06, "num_tokens": 855791.0, "completions/mean_length": 134.25, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 80.28572082519531, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.865929901599884, "rewards/meter/std": 0.3411285877227783, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.865929901599884, "rewards/total_composite/std": 0.3411285877227783, "reward": 0.865929901599884, "reward_std": 0.3411285877227783, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07304941862821579, "sampling/sampling_logp_difference/max": 1.6561942100524902, "sampling/importance_sampling_ratio/min": 0.19086399674415588, "sampling/importance_sampling_ratio/mean": 1.0096566677093506, "sampling/importance_sampling_ratio/max": 1.667251706123352, "entropy": 0.4520561061799526, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/high_mean": 0.04474749346263707, "clip_ratio/high_max": 0.04474749346263707, "clip_ratio/region_mean": 0.056111130164936185, "reward_total_mean": 0.865929901599884, "reward_meter_mean": 0.865929901599884, "reward_meter_std": 0.3411285877227783, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.865929901599884, "reward_total_composite_std": 0.3411285877227783, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 397.0} {"timestamp_utc": "2026-04-11T20:11:15Z", "mode": "train", "global_step": 398, "epoch": 0.015369168983626814, "loss": 0.1437, "grad_norm": 4.518268585205078, "learning_rate": 8.796969696969698e-06, "num_tokens": 857766.0, "completions/mean_length": 89.875, "completions/min_length": 75.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.875, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9967172145843506, "rewards/meter/std": 0.003067870857194066, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9344407320022583, "rewards/total_composite/std": 0.17628978192806244, "reward": 0.9344407320022583, "reward_std": 0.17628978192806244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05183165520429611, "sampling/sampling_logp_difference/max": 1.3592205047607422, "sampling/importance_sampling_ratio/min": 0.2568609118461609, "sampling/importance_sampling_ratio/mean": 1.002355694770813, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38309746980667114, "clip_ratio/low_mean": 0.005040322430431843, "clip_ratio/low_min": 0.005040322430431843, "clip_ratio/high_mean": 0.04711233067791909, "clip_ratio/high_max": 0.04711233067791909, "clip_ratio/region_mean": 0.05215265310835093, "reward_total_mean": 0.9344407320022583, "reward_meter_mean": 0.9967172145843506, "reward_meter_std": 0.003067870857194066, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9344407320022583, "reward_total_composite_std": 0.17628978192806244, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 398.0} {"timestamp_utc": "2026-04-11T20:11:20Z", "mode": "train", "global_step": 399, "epoch": 0.015407784986098239, "loss": 0.0428, "grad_norm": 10.965712547302246, "learning_rate": 8.793939393939395e-06, "num_tokens": 859500.0, "completions/mean_length": 53.75, "completions/min_length": 47.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.75, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.37250423431396484, "rewards/meter/std": 0.42634570598602295, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.26512396335601807, "rewards/total_composite/std": 0.3931903541088104, "reward": 0.26512396335601807, "reward_std": 0.39319032430648804, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0987580195069313, "sampling/sampling_logp_difference/max": 2.4608466625213623, "sampling/importance_sampling_ratio/min": 0.08536264300346375, "sampling/importance_sampling_ratio/mean": 1.001451015472412, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45951317623257637, "clip_ratio/low_mean": 0.03256993810646236, "clip_ratio/low_min": 0.03256993810646236, "clip_ratio/high_mean": 0.03413120610639453, "clip_ratio/high_max": 0.03413120610639453, "clip_ratio/region_mean": 0.06670114421285689, "reward_total_mean": 0.26512396335601807, "reward_meter_mean": 0.37250423431396484, "reward_meter_std": 0.42634570598602295, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.26512396335601807, "reward_total_composite_std": 0.3931903541088104, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 399.0} {"timestamp_utc": "2026-04-11T20:11:30Z", "mode": "train", "global_step": 400, "epoch": 0.015446400988569663, "loss": -0.2368, "grad_norm": 0.6731557250022888, "learning_rate": 8.790909090909092e-06, "num_tokens": 862030.0, "completions/mean_length": 196.25, "completions/min_length": 148.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 151.1428680419922, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.9850339889526367, "rewards/meter/std": 0.02169320359826088, "rewards/count_adherence/mean": 0.65625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6456302404403687, "rewards/total_composite/std": 0.26136812567710876, "reward": 0.6456302404403687, "reward_std": 0.2613680958747864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020483043044805527, "sampling/sampling_logp_difference/max": 1.6783409118652344, "sampling/importance_sampling_ratio/min": 0.18668344616889954, "sampling/importance_sampling_ratio/mean": 1.002915382385254, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0809963047504425, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.011500443739350885, "clip_ratio/high_max": 0.011500443739350885, "clip_ratio/region_mean": 0.011500443739350885, "reward_total_mean": 0.6456302404403687, "reward_meter_mean": 0.9850339889526367, "reward_meter_std": 0.02169320359826088, "reward_count_adherence_mean": 0.65625, "reward_count_adherence_std": 0.2651650309562683, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6456302404403687, "reward_total_composite_std": 0.26136812567710876, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 400.0} {"timestamp_utc": "2026-04-11T20:13:06Z", "mode": "eval", "global_step": 400, "epoch": 0.015446400988569663, "eval_loss": NaN, "eval_runtime": 96.6201, "eval_samples_per_second": 1.076, "eval_steps_per_second": 0.135, "eval_num_tokens": 862030.0, "eval_completions/mean_length": 344.08653846153845, "eval_completions/min_length": 85.23076923076923, "eval_completions/max_length": 512.0, "eval_completions/clipped_ratio": 0.41346153846153844, "eval_completions/mean_terminated_length": 226.2978057861328, "eval_completions/min_terminated_length": 85.23076923076923, "eval_completions/max_terminated_length": 400.53846153846155, "eval_rewards/meter/mean": 0.6171861015833341, "eval_rewards/meter/std": 0.42123945630513704, "eval_rewards/count_adherence/mean": 0.6621637413134942, "eval_rewards/count_adherence/std": 0.3351786213998611, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/total_composite/mean": 0.4161693981060615, "eval_rewards/total_composite/std": 0.378701776266098, "eval_reward": 0.4161693981060615, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.018011176170637973, "eval_sampling/sampling_logp_difference/max": 0.9377481570610633, "eval_sampling/importance_sampling_ratio/min": 0.40342745643395644, "eval_sampling/importance_sampling_ratio/mean": 1.0045312092854426, "eval_sampling/importance_sampling_ratio/max": 1.3838598086283758, "eval_entropy": 0.19937350027836287, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4161693981060615, "eval_reward_meter_mean": 0.6171861015833341, "eval_reward_meter_std": 0.42123945630513704, "eval_reward_count_adherence_mean": 0.6621637413134942, "eval_reward_count_adherence_std": 0.3351786213998611, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_total_composite_mean": 0.4161693981060615, "eval_reward_total_composite_std": 0.378701776266098, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 400.0} {"timestamp_utc": "2026-04-11T20:13:14Z", "mode": "train", "global_step": 401, "epoch": 0.015485016991041087, "loss": 0.0529, "grad_norm": 6.882446765899658, "learning_rate": 8.787878787878788e-06, "num_tokens": 863793.0, "completions/mean_length": 70.375, "completions/min_length": 64.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.5120395421981812, "rewards/meter/std": 0.39659610390663147, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5120395421981812, "rewards/total_composite/std": 0.39659610390663147, "reward": 0.5120395421981812, "reward_std": 0.39659610390663147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.079770028591156, "sampling/sampling_logp_difference/max": 0.8284692764282227, "sampling/importance_sampling_ratio/min": 0.43671727180480957, "sampling/importance_sampling_ratio/mean": 1.0185896158218384, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6498333364725113, "clip_ratio/low_mean": 0.04286885913461447, "clip_ratio/low_min": 0.04286885913461447, "clip_ratio/high_mean": 0.011423650896176696, "clip_ratio/high_max": 0.011423650896176696, "clip_ratio/region_mean": 0.05429251003079116, "reward_total_mean": 0.5120395421981812, "reward_meter_mean": 0.5120395421981812, "reward_meter_std": 0.39659610390663147, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5120395421981812, "reward_total_composite_std": 0.39659610390663147, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 401.0} {"timestamp_utc": "2026-04-11T20:13:20Z", "mode": "train", "global_step": 402, "epoch": 0.01552363299351251, "loss": -0.0316, "grad_norm": 3.843118190765381, "learning_rate": 8.784848484848487e-06, "num_tokens": 865768.0, "completions/mean_length": 90.875, "completions/min_length": 76.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.875, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9842378497123718, "rewards/meter/std": 0.0201681200414896, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9842378497123718, "rewards/total_composite/std": 0.0201681200414896, "reward": 0.9842378497123718, "reward_std": 0.020168133080005646, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03747996687889099, "sampling/sampling_logp_difference/max": 1.155746579170227, "sampling/importance_sampling_ratio/min": 0.31482240557670593, "sampling/importance_sampling_ratio/mean": 1.0152220726013184, "sampling/importance_sampling_ratio/max": 1.9674099683761597, "entropy": 0.2551422342658043, "clip_ratio/low_mean": 0.013676032423973083, "clip_ratio/low_min": 0.013676032423973083, "clip_ratio/high_mean": 0.01652618101797998, "clip_ratio/high_max": 0.01652618101797998, "clip_ratio/region_mean": 0.030202213441953063, "reward_total_mean": 0.9842378497123718, "reward_meter_mean": 0.9842378497123718, "reward_meter_std": 0.0201681200414896, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9842378497123718, "reward_total_composite_std": 0.0201681200414896, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 402.0} {"timestamp_utc": "2026-04-11T20:13:30Z", "mode": "train", "global_step": 403, "epoch": 0.015562248995983935, "loss": -0.113, "grad_norm": 1.8304016590118408, "learning_rate": 8.781818181818182e-06, "num_tokens": 868356.0, "completions/mean_length": 217.5, "completions/min_length": 140.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 175.42857360839844, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 267.0, "rewards/meter/mean": 0.9523094892501831, "rewards/meter/std": 0.1186181977391243, "rewards/count_adherence/mean": 0.5833333730697632, "rewards/count_adherence/std": 0.29546844959259033, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5801796913146973, "rewards/total_composite/std": 0.29439741373062134, "reward": 0.5801796913146973, "reward_std": 0.29439738392829895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.043538108468055725, "sampling/sampling_logp_difference/max": 1.0842866897583008, "sampling/importance_sampling_ratio/min": 0.33814290165901184, "sampling/importance_sampling_ratio/mean": 1.0064983367919922, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3300076089799404, "clip_ratio/low_mean": 0.000936329597607255, "clip_ratio/low_min": 0.000936329597607255, "clip_ratio/high_mean": 0.028843345178756863, "clip_ratio/high_max": 0.028843345178756863, "clip_ratio/region_mean": 0.029779674776364118, "reward_total_mean": 0.5801796913146973, "reward_meter_mean": 0.9523094892501831, "reward_meter_std": 0.1186181977391243, "reward_count_adherence_mean": 0.5833333730697632, "reward_count_adherence_std": 0.29546844959259033, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5801796913146973, "reward_total_composite_std": 0.29439741373062134, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 403.0} {"timestamp_utc": "2026-04-11T20:13:39Z", "mode": "train", "global_step": 404, "epoch": 0.015600864998455359, "loss": -0.1182, "grad_norm": 1.5644749402999878, "learning_rate": 8.77878787878788e-06, "num_tokens": 870166.0, "completions/mean_length": 187.25, "completions/min_length": 75.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 79.0, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.7447091341018677, "rewards/meter/std": 0.3070685863494873, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5707218647003174, "rewards/total_composite/std": 0.45596015453338623, "reward": 0.5707218647003174, "reward_std": 0.45596015453338623, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06043194606900215, "sampling/sampling_logp_difference/max": 1.4539172649383545, "sampling/importance_sampling_ratio/min": 0.2336532175540924, "sampling/importance_sampling_ratio/mean": 1.0064294338226318, "sampling/importance_sampling_ratio/max": 1.7909971475601196, "entropy": 0.26876694336533546, "clip_ratio/low_mean": 0.005681818351149559, "clip_ratio/low_min": 0.005681818351149559, "clip_ratio/high_mean": 0.029422825085930526, "clip_ratio/high_max": 0.029422825085930526, "clip_ratio/region_mean": 0.035104643437080085, "reward_total_mean": 0.5707218647003174, "reward_meter_mean": 0.7447091341018677, "reward_meter_std": 0.3070685863494873, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5707218647003174, "reward_total_composite_std": 0.45596015453338623, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 404.0} {"timestamp_utc": "2026-04-11T20:13:50Z", "mode": "train", "global_step": 405, "epoch": 0.015639481000926783, "loss": 0.181, "grad_norm": 1.1297802925109863, "learning_rate": 8.775757575757577e-06, "num_tokens": 872988.0, "completions/mean_length": 300.75, "completions/min_length": 193.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 230.33334350585938, "completions/min_terminated_length": 193.0, "completions/max_terminated_length": 256.0, "rewards/meter/mean": 0.37604400515556335, "rewards/meter/std": 0.43697869777679443, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.22160132229328156, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.23586036264896393, "rewards/total_composite/std": 0.3007969558238983, "reward": 0.23586036264896393, "reward_std": 0.3007969558238983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023270348086953163, "sampling/sampling_logp_difference/max": 1.2483367919921875, "sampling/importance_sampling_ratio/min": 0.2869817018508911, "sampling/importance_sampling_ratio/mean": 1.0042189359664917, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12349590379744768, "clip_ratio/low_mean": 0.012407735688611865, "clip_ratio/low_min": 0.012407735688611865, "clip_ratio/high_mean": 0.0038860102649778128, "clip_ratio/high_max": 0.0038860102649778128, "clip_ratio/region_mean": 0.016293745953589678, "reward_total_mean": 0.23586036264896393, "reward_meter_mean": 0.37604400515556335, "reward_meter_std": 0.43697869777679443, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.22160132229328156, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.23586036264896393, "reward_total_composite_std": 0.3007969558238983, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 405.0} {"timestamp_utc": "2026-04-11T20:13:55Z", "mode": "train", "global_step": 406, "epoch": 0.015678097003398207, "loss": 0.0324, "grad_norm": 4.1059088706970215, "learning_rate": 8.772727272727274e-06, "num_tokens": 874753.0, "completions/mean_length": 62.625, "completions/min_length": 60.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.625, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9913897514343262, "rewards/meter/std": 0.003956401254981756, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9913897514343262, "rewards/total_composite/std": 0.003956401254981756, "reward": 0.9913897514343262, "reward_std": 0.003956401254981756, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027037320658564568, "sampling/sampling_logp_difference/max": 0.9925615787506104, "sampling/importance_sampling_ratio/min": 0.4850950837135315, "sampling/importance_sampling_ratio/mean": 1.0122790336608887, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21097453031688929, "clip_ratio/low_mean": 0.011723854579031467, "clip_ratio/low_min": 0.011723854579031467, "clip_ratio/high_mean": 0.0061827958561480045, "clip_ratio/high_max": 0.0061827958561480045, "clip_ratio/region_mean": 0.017906650435179472, "reward_total_mean": 0.9913897514343262, "reward_meter_mean": 0.9913897514343262, "reward_meter_std": 0.003956401254981756, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9913897514343262, "reward_total_composite_std": 0.003956401254981756, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 406.0} {"timestamp_utc": "2026-04-11T20:14:01Z", "mode": "train", "global_step": 407, "epoch": 0.01571671300586963, "loss": 0.0382, "grad_norm": 3.022797107696533, "learning_rate": 8.76969696969697e-06, "num_tokens": 877196.0, "completions/mean_length": 138.375, "completions/min_length": 121.0, "completions/max_length": 168.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.375, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 168.0, "rewards/meter/mean": 0.6999959945678711, "rewards/meter/std": 0.3435265123844147, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.46666404604911804, "rewards/total_composite/std": 0.22901767492294312, "reward": 0.46666404604911804, "reward_std": 0.22901766002178192, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022317711263895035, "sampling/sampling_logp_difference/max": 1.4253215789794922, "sampling/importance_sampling_ratio/min": 0.24043114483356476, "sampling/importance_sampling_ratio/mean": 1.0028660297393799, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1530680824071169, "clip_ratio/low_mean": 0.011579734331462532, "clip_ratio/low_min": 0.011579734331462532, "clip_ratio/high_mean": 0.016171834780834615, "clip_ratio/high_max": 0.016171834780834615, "clip_ratio/region_mean": 0.027751569112297148, "reward_total_mean": 0.46666404604911804, "reward_meter_mean": 0.6999959945678711, "reward_meter_std": 0.3435265123844147, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.46666404604911804, "reward_total_composite_std": 0.22901767492294312, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 407.0} {"timestamp_utc": "2026-04-11T20:14:06Z", "mode": "train", "global_step": 408, "epoch": 0.015755329008341055, "loss": 0.1037, "grad_norm": 10.218791007995605, "learning_rate": 8.766666666666669e-06, "num_tokens": 878828.0, "completions/mean_length": 59.0, "completions/min_length": 40.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.6509081721305847, "rewards/meter/std": 0.4415837228298187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6509081721305847, "rewards/total_composite/std": 0.4415837228298187, "reward": 0.6509081721305847, "reward_std": 0.44158369302749634, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11866805702447891, "sampling/sampling_logp_difference/max": 2.771667718887329, "sampling/importance_sampling_ratio/min": 0.06255759298801422, "sampling/importance_sampling_ratio/mean": 1.02310311794281, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9289319105446339, "clip_ratio/low_mean": 0.036312646232545376, "clip_ratio/low_min": 0.036312646232545376, "clip_ratio/high_mean": 0.08028767257928848, "clip_ratio/high_max": 0.08028767257928848, "clip_ratio/region_mean": 0.11660031881183386, "reward_total_mean": 0.6509081721305847, "reward_meter_mean": 0.6509081721305847, "reward_meter_std": 0.4415837228298187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6509081721305847, "reward_total_composite_std": 0.4415837228298187, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 408.0} {"timestamp_utc": "2026-04-11T20:14:13Z", "mode": "train", "global_step": 409, "epoch": 0.01579394501081248, "loss": 0.0206, "grad_norm": 1.702805995941162, "learning_rate": 8.763636363636364e-06, "num_tokens": 881641.0, "completions/mean_length": 175.625, "completions/min_length": 159.0, "completions/max_length": 216.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 175.625, "completions/min_terminated_length": 159.0, "completions/max_terminated_length": 216.0, "rewards/meter/mean": 0.9501590132713318, "rewards/meter/std": 0.11174096912145615, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7126193046569824, "rewards/total_composite/std": 0.08380572497844696, "reward": 0.7126193046569824, "reward_std": 0.08380571752786636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013060882687568665, "sampling/sampling_logp_difference/max": 0.9655053615570068, "sampling/importance_sampling_ratio/min": 0.38079074025154114, "sampling/importance_sampling_ratio/mean": 1.001296043395996, "sampling/importance_sampling_ratio/max": 1.7096338272094727, "entropy": 0.07792735006660223, "clip_ratio/low_mean": 0.0006944444612599909, "clip_ratio/low_min": 0.0006944444612599909, "clip_ratio/high_mean": 0.009341755299828947, "clip_ratio/high_max": 0.009341755299828947, "clip_ratio/region_mean": 0.010036199761088938, "reward_total_mean": 0.7126193046569824, "reward_meter_mean": 0.9501590132713318, "reward_meter_std": 0.11174096912145615, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7126193046569824, "reward_total_composite_std": 0.08380572497844696, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 409.0} {"timestamp_utc": "2026-04-11T20:14:20Z", "mode": "train", "global_step": 410, "epoch": 0.015832561013283904, "loss": -0.0088, "grad_norm": 2.0530569553375244, "learning_rate": 8.760606060606061e-06, "num_tokens": 884542.0, "completions/mean_length": 197.625, "completions/min_length": 182.0, "completions/max_length": 244.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 197.625, "completions/min_terminated_length": 182.0, "completions/max_terminated_length": 244.0, "rewards/meter/mean": 0.8907101154327393, "rewards/meter/std": 0.20771878957748413, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7310122847557068, "rewards/total_composite/std": 0.1586739867925644, "reward": 0.7310122847557068, "reward_std": 0.1586739867925644, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017095215618610382, "sampling/sampling_logp_difference/max": 4.624732971191406, "sampling/importance_sampling_ratio/min": 0.009806273505091667, "sampling/importance_sampling_ratio/mean": 0.9990739822387695, "sampling/importance_sampling_ratio/max": 1.6074455976486206, "entropy": 0.0969978254288435, "clip_ratio/low_mean": 0.003342245938256383, "clip_ratio/low_min": 0.003342245938256383, "clip_ratio/high_mean": 0.012198542011901736, "clip_ratio/high_max": 0.012198542011901736, "clip_ratio/region_mean": 0.01554078795015812, "reward_total_mean": 0.7310122847557068, "reward_meter_mean": 0.8907101154327393, "reward_meter_std": 0.20771878957748413, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7310122847557068, "reward_total_composite_std": 0.1586739867925644, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 410.0} {"timestamp_utc": "2026-04-11T20:14:26Z", "mode": "train", "global_step": 411, "epoch": 0.015871177015755328, "loss": 0.1971, "grad_norm": 4.408254623413086, "learning_rate": 8.757575757575759e-06, "num_tokens": 886841.0, "completions/mean_length": 105.375, "completions/min_length": 83.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.375, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.9555299282073975, "rewards/meter/std": 0.10529791563749313, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8437220454216003, "rewards/total_composite/std": 0.2139330953359604, "reward": 0.8437220454216003, "reward_std": 0.2139330804347992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03925402835011482, "sampling/sampling_logp_difference/max": 1.2860984802246094, "sampling/importance_sampling_ratio/min": 0.2763468325138092, "sampling/importance_sampling_ratio/mean": 1.0038602352142334, "sampling/importance_sampling_ratio/max": 1.7728590965270996, "entropy": 0.23645233362913132, "clip_ratio/low_mean": 0.005724609247408807, "clip_ratio/low_min": 0.005724609247408807, "clip_ratio/high_mean": 0.02881905622780323, "clip_ratio/high_max": 0.02881905622780323, "clip_ratio/region_mean": 0.03454366547521204, "reward_total_mean": 0.8437220454216003, "reward_meter_mean": 0.9555299282073975, "reward_meter_std": 0.10529791563749313, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8437220454216003, "reward_total_composite_std": 0.2139330953359604, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 411.0} {"timestamp_utc": "2026-04-11T20:14:37Z", "mode": "train", "global_step": 412, "epoch": 0.015909793018226752, "loss": -0.2932, "grad_norm": 0.3299441337585449, "learning_rate": 8.754545454545456e-06, "num_tokens": 891477.0, "completions/mean_length": 413.5, "completions/min_length": 378.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 399.4285888671875, "completions/min_terminated_length": 378.0, "completions/max_terminated_length": 441.0, "rewards/meter/mean": 0.8727073669433594, "rewards/meter/std": 0.34727609157562256, "rewards/count_adherence/mean": 0.8194444179534912, "rewards/count_adherence/std": 0.29057809710502625, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8021494150161743, "rewards/total_composite/std": 0.3275650441646576, "reward": 0.8021494150161743, "reward_std": 0.3275650143623352, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007050658110529184, "sampling/sampling_logp_difference/max": 1.4844205379486084, "sampling/importance_sampling_ratio/min": 0.22663362324237823, "sampling/importance_sampling_ratio/mean": 0.9996734857559204, "sampling/importance_sampling_ratio/max": 1.6943782567977905, "entropy": 0.03046043962240219, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005883037491003051, "clip_ratio/high_max": 0.005883037491003051, "clip_ratio/region_mean": 0.005883037491003051, "reward_total_mean": 0.8021494150161743, "reward_meter_mean": 0.8727073669433594, "reward_meter_std": 0.34727609157562256, "reward_count_adherence_mean": 0.8194444179534912, "reward_count_adherence_std": 0.29057809710502625, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8021494150161743, "reward_total_composite_std": 0.3275650441646576, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 412.0} {"timestamp_utc": "2026-04-11T20:14:46Z", "mode": "train", "global_step": 413, "epoch": 0.015948409020698176, "loss": 0.0018, "grad_norm": 2.0739166736602783, "learning_rate": 8.751515151515151e-06, "num_tokens": 892998.0, "completions/mean_length": 167.125, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 52.16666793823242, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.5100301504135132, "rewards/meter/std": 0.5075311064720154, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.5175492167472839, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5062292814254761, "rewards/total_composite/std": 0.5117626786231995, "reward": 0.5062292814254761, "reward_std": 0.5117626786231995, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06872761994600296, "sampling/sampling_logp_difference/max": 0.7841591835021973, "sampling/importance_sampling_ratio/min": 0.5460512638092041, "sampling/importance_sampling_ratio/mean": 1.01053786277771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5243904180824757, "clip_ratio/low_mean": 0.012824675533920527, "clip_ratio/low_min": 0.012824675533920527, "clip_ratio/high_mean": 0.05882604233920574, "clip_ratio/high_max": 0.05882604233920574, "clip_ratio/region_mean": 0.07165071787312627, "reward_total_mean": 0.5062292814254761, "reward_meter_mean": 0.5100301504135132, "reward_meter_std": 0.5075311064720154, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.5175492167472839, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5062292814254761, "reward_total_composite_std": 0.5117626786231995, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 413.0} {"timestamp_utc": "2026-04-11T20:14:51Z", "mode": "train", "global_step": 414, "epoch": 0.0159870250231696, "loss": 0.0655, "grad_norm": 10.703254699707031, "learning_rate": 8.748484848484849e-06, "num_tokens": 894651.0, "completions/mean_length": 57.625, "completions/min_length": 52.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.625, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9551189541816711, "rewards/meter/std": 0.09812162071466446, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9551189541816711, "rewards/total_composite/std": 0.09812162071466446, "reward": 0.9551189541816711, "reward_std": 0.09812159836292267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07875344157218933, "sampling/sampling_logp_difference/max": 1.2454400062561035, "sampling/importance_sampling_ratio/min": 0.28781425952911377, "sampling/importance_sampling_ratio/mean": 1.0104643106460571, "sampling/importance_sampling_ratio/max": 1.675656795501709, "entropy": 0.582982636988163, "clip_ratio/low_mean": 0.009469697251915932, "clip_ratio/low_min": 0.009469697251915932, "clip_ratio/high_mean": 0.06984005169942975, "clip_ratio/high_max": 0.06984005169942975, "clip_ratio/region_mean": 0.07930974895134568, "reward_total_mean": 0.9551189541816711, "reward_meter_mean": 0.9551189541816711, "reward_meter_std": 0.09812162071466446, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9551189541816711, "reward_total_composite_std": 0.09812162071466446, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 414.0} {"timestamp_utc": "2026-04-11T20:15:01Z", "mode": "train", "global_step": 415, "epoch": 0.016025641025641024, "loss": 0.0492, "grad_norm": 4.649420261383057, "learning_rate": 8.745454545454546e-06, "num_tokens": 896156.0, "completions/mean_length": 112.125, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 55.000003814697266, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.2886698246002197, "rewards/meter/std": 0.40278124809265137, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.28754958510398865, "rewards/total_composite/std": 0.40368756651878357, "reward": 0.28754958510398865, "reward_std": 0.40368756651878357, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10733786970376968, "sampling/sampling_logp_difference/max": 1.5200672149658203, "sampling/importance_sampling_ratio/min": 0.2186972051858902, "sampling/importance_sampling_ratio/mean": 1.0061570405960083, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6476233620196581, "clip_ratio/low_mean": 0.03434053680393845, "clip_ratio/low_min": 0.03434053680393845, "clip_ratio/high_mean": 0.04313966212794185, "clip_ratio/high_max": 0.04313966212794185, "clip_ratio/region_mean": 0.0774801989318803, "reward_total_mean": 0.28754958510398865, "reward_meter_mean": 0.2886698246002197, "reward_meter_std": 0.40278124809265137, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.28754958510398865, "reward_total_composite_std": 0.40368756651878357, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 415.0} {"timestamp_utc": "2026-04-11T20:15:06Z", "mode": "train", "global_step": 416, "epoch": 0.01606425702811245, "loss": 0.0358, "grad_norm": 7.304360866546631, "learning_rate": 8.742424242424243e-06, "num_tokens": 897867.0, "completions/mean_length": 57.875, "completions/min_length": 51.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9717522859573364, "rewards/meter/std": 0.027185985818505287, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9717522859573364, "rewards/total_composite/std": 0.027185985818505287, "reward": 0.9717522859573364, "reward_std": 0.027185987681150436, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09120234102010727, "sampling/sampling_logp_difference/max": 1.6343998908996582, "sampling/importance_sampling_ratio/min": 0.19506938755512238, "sampling/importance_sampling_ratio/mean": 1.0156903266906738, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6767406091094017, "clip_ratio/low_mean": 0.03538359794765711, "clip_ratio/low_min": 0.03538359794765711, "clip_ratio/high_mean": 0.062005657935515046, "clip_ratio/high_max": 0.062005657935515046, "clip_ratio/region_mean": 0.09738925588317215, "reward_total_mean": 0.9717522859573364, "reward_meter_mean": 0.9717522859573364, "reward_meter_std": 0.027185985818505287, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9717522859573364, "reward_total_composite_std": 0.027185985818505287, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 416.0} {"timestamp_utc": "2026-04-11T20:15:14Z", "mode": "train", "global_step": 417, "epoch": 0.016102873030583872, "loss": 0.1247, "grad_norm": 1.8169100284576416, "learning_rate": 8.73939393939394e-06, "num_tokens": 901164.0, "completions/mean_length": 227.125, "completions/min_length": 102.0, "completions/max_length": 344.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 227.125, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 344.0, "rewards/meter/mean": 0.6192874312400818, "rewards/meter/std": 0.45972275733947754, "rewards/count_adherence/mean": 0.925000011920929, "rewards/count_adherence/std": 0.14880475401878357, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5552949905395508, "rewards/total_composite/std": 0.43653252720832825, "reward": 0.5552949905395508, "reward_std": 0.43653252720832825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02308918908238411, "sampling/sampling_logp_difference/max": 1.3905742168426514, "sampling/importance_sampling_ratio/min": 0.27433061599731445, "sampling/importance_sampling_ratio/mean": 1.0012174844741821, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2837667972780764, "clip_ratio/low_mean": 0.002363372186664492, "clip_ratio/low_min": 0.002363372186664492, "clip_ratio/high_mean": 0.021424076927360147, "clip_ratio/high_max": 0.021424076927360147, "clip_ratio/region_mean": 0.02378744911402464, "reward_total_mean": 0.5552949905395508, "reward_meter_mean": 0.6192874312400818, "reward_meter_std": 0.45972275733947754, "reward_count_adherence_mean": 0.925000011920929, "reward_count_adherence_std": 0.14880475401878357, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5552949905395508, "reward_total_composite_std": 0.43653252720832825, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 417.0} {"timestamp_utc": "2026-04-11T20:15:19Z", "mode": "train", "global_step": 418, "epoch": 0.016141489033055297, "loss": -0.0317, "grad_norm": 21.298921585083008, "learning_rate": 8.736363636363638e-06, "num_tokens": 902909.0, "completions/mean_length": 50.125, "completions/min_length": 37.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.125, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.32849666476249695, "rewards/meter/std": 0.32036224007606506, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.32849666476249695, "rewards/total_composite/std": 0.32036224007606506, "reward": 0.32849666476249695, "reward_std": 0.32036224007606506, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13514763116836548, "sampling/sampling_logp_difference/max": 2.284748077392578, "sampling/importance_sampling_ratio/min": 0.14722536504268646, "sampling/importance_sampling_ratio/mean": 1.0049675703048706, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8871809840202332, "clip_ratio/low_mean": 0.11292503494769335, "clip_ratio/low_min": 0.11292503494769335, "clip_ratio/high_mean": 0.03858941700309515, "clip_ratio/high_max": 0.03858941700309515, "clip_ratio/region_mean": 0.1515144519507885, "reward_total_mean": 0.32849666476249695, "reward_meter_mean": 0.32849666476249695, "reward_meter_std": 0.32036224007606506, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.32849666476249695, "reward_total_composite_std": 0.32036224007606506, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 418.0} {"timestamp_utc": "2026-04-11T20:15:24Z", "mode": "train", "global_step": 419, "epoch": 0.01618010503552672, "loss": 0.0702, "grad_norm": 6.602462291717529, "learning_rate": 8.733333333333333e-06, "num_tokens": 904739.0, "completions/mean_length": 73.75, "completions/min_length": 60.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.75, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.799231767654419, "rewards/meter/std": 0.3458506464958191, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.799231767654419, "rewards/total_composite/std": 0.3458506464958191, "reward": 0.799231767654419, "reward_std": 0.3458506166934967, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07469379901885986, "sampling/sampling_logp_difference/max": 1.0570578575134277, "sampling/importance_sampling_ratio/min": 0.34747663140296936, "sampling/importance_sampling_ratio/mean": 0.9987403750419617, "sampling/importance_sampling_ratio/max": 1.8707276582717896, "entropy": 0.5439852885901928, "clip_ratio/low_mean": 0.00957037159241736, "clip_ratio/low_min": 0.00957037159241736, "clip_ratio/high_mean": 0.05910295504145324, "clip_ratio/high_max": 0.05910295504145324, "clip_ratio/region_mean": 0.0686733266338706, "reward_total_mean": 0.799231767654419, "reward_meter_mean": 0.799231767654419, "reward_meter_std": 0.3458506464958191, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.799231767654419, "reward_total_composite_std": 0.3458506464958191, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 419.0} {"timestamp_utc": "2026-04-11T20:15:29Z", "mode": "train", "global_step": 420, "epoch": 0.016218721037998145, "loss": 0.0655, "grad_norm": 4.019112586975098, "learning_rate": 8.73030303030303e-06, "num_tokens": 906465.0, "completions/mean_length": 72.75, "completions/min_length": 64.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9897938966751099, "rewards/meter/std": 0.00681549496948719, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9897938966751099, "rewards/total_composite/std": 0.00681549496948719, "reward": 0.9897938966751099, "reward_std": 0.006815491709858179, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030092770233750343, "sampling/sampling_logp_difference/max": 2.5420010089874268, "sampling/importance_sampling_ratio/min": 0.07870874553918839, "sampling/importance_sampling_ratio/mean": 1.0009511709213257, "sampling/importance_sampling_ratio/max": 1.771165132522583, "entropy": 0.13834022916853428, "clip_ratio/low_mean": 0.006358225247822702, "clip_ratio/low_min": 0.006358225247822702, "clip_ratio/high_mean": 0.014262907905504107, "clip_ratio/high_max": 0.014262907905504107, "clip_ratio/region_mean": 0.02062113315332681, "reward_total_mean": 0.9897938966751099, "reward_meter_mean": 0.9897938966751099, "reward_meter_std": 0.00681549496948719, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9897938966751099, "reward_total_composite_std": 0.00681549496948719, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 420.0} {"timestamp_utc": "2026-04-11T20:15:39Z", "mode": "train", "global_step": 421, "epoch": 0.016257337040469572, "loss": -0.119, "grad_norm": 2.0794014930725098, "learning_rate": 8.727272727272728e-06, "num_tokens": 908794.0, "completions/mean_length": 189.125, "completions/min_length": 136.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 143.0, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.6905167698860168, "rewards/meter/std": 0.3626475930213928, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6646036505699158, "rewards/total_composite/std": 0.4017622768878937, "reward": 0.6646036505699158, "reward_std": 0.4017622768878937, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02924906462430954, "sampling/sampling_logp_difference/max": 0.9549951553344727, "sampling/importance_sampling_ratio/min": 0.39857566356658936, "sampling/importance_sampling_ratio/mean": 1.0038185119628906, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14536871574819088, "clip_ratio/low_mean": 0.005128088872879744, "clip_ratio/low_min": 0.005128088872879744, "clip_ratio/high_mean": 0.016083280788734555, "clip_ratio/high_max": 0.016083280788734555, "clip_ratio/region_mean": 0.0212113696616143, "reward_total_mean": 0.6646036505699158, "reward_meter_mean": 0.6905167698860168, "reward_meter_std": 0.3626475930213928, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6646036505699158, "reward_total_composite_std": 0.4017622768878937, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 421.0} {"timestamp_utc": "2026-04-11T20:15:45Z", "mode": "train", "global_step": 422, "epoch": 0.016295953042940996, "loss": -0.026, "grad_norm": 4.160637855529785, "learning_rate": 8.724242424242425e-06, "num_tokens": 911117.0, "completions/mean_length": 121.375, "completions/min_length": 109.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.375, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.9373536109924316, "rewards/meter/std": 0.09886158257722855, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9373536109924316, "rewards/total_composite/std": 0.09886158257722855, "reward": 0.9373536109924316, "reward_std": 0.09886158257722855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05337180942296982, "sampling/sampling_logp_difference/max": 1.349278450012207, "sampling/importance_sampling_ratio/min": 0.25942739844322205, "sampling/importance_sampling_ratio/mean": 1.011407732963562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49255986511707306, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/high_mean": 0.04184396180789918, "clip_ratio/high_max": 0.04184396180789918, "clip_ratio/region_mean": 0.046229926752857864, "reward_total_mean": 0.9373536109924316, "reward_meter_mean": 0.9373536109924316, "reward_meter_std": 0.09886158257722855, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9373536109924316, "reward_total_composite_std": 0.09886158257722855, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 422.0} {"timestamp_utc": "2026-04-11T20:15:50Z", "mode": "train", "global_step": 423, "epoch": 0.01633456904541242, "loss": 0.0591, "grad_norm": 13.163901329040527, "learning_rate": 8.72121212121212e-06, "num_tokens": 912910.0, "completions/mean_length": 66.125, "completions/min_length": 60.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9770172834396362, "rewards/meter/std": 0.024198686704039574, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9770172834396362, "rewards/total_composite/std": 0.024198686704039574, "reward": 0.9770172834396362, "reward_std": 0.024198684841394424, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04488004371523857, "sampling/sampling_logp_difference/max": 3.4570813179016113, "sampling/importance_sampling_ratio/min": 0.03152162954211235, "sampling/importance_sampling_ratio/mean": 1.003563642501831, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16816303879022598, "clip_ratio/low_mean": 0.018862260156311095, "clip_ratio/low_min": 0.018862260156311095, "clip_ratio/high_mean": 0.01431451621465385, "clip_ratio/high_max": 0.01431451621465385, "clip_ratio/region_mean": 0.033176776370964944, "reward_total_mean": 0.9770172834396362, "reward_meter_mean": 0.9770172834396362, "reward_meter_std": 0.024198686704039574, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9770172834396362, "reward_total_composite_std": 0.024198686704039574, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 423.0} {"timestamp_utc": "2026-04-11T20:15:55Z", "mode": "train", "global_step": 424, "epoch": 0.016373185047883845, "loss": 0.0378, "grad_norm": 6.008713722229004, "learning_rate": 8.71818181818182e-06, "num_tokens": 914750.0, "completions/mean_length": 78.0, "completions/min_length": 55.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.7533670663833618, "rewards/meter/std": 0.23079943656921387, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7533670663833618, "rewards/total_composite/std": 0.23079943656921387, "reward": 0.7533670663833618, "reward_std": 0.23079942166805267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0553349107503891, "sampling/sampling_logp_difference/max": 4.345860958099365, "sampling/importance_sampling_ratio/min": 0.012960345484316349, "sampling/importance_sampling_ratio/mean": 0.9994415044784546, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.281008530408144, "clip_ratio/low_mean": 0.013393073342740536, "clip_ratio/low_min": 0.013393073342740536, "clip_ratio/high_mean": 0.04446423542685807, "clip_ratio/high_max": 0.04446423542685807, "clip_ratio/region_mean": 0.0578573087695986, "reward_total_mean": 0.7533670663833618, "reward_meter_mean": 0.7533670663833618, "reward_meter_std": 0.23079943656921387, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7533670663833618, "reward_total_composite_std": 0.23079943656921387, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 424.0} {"timestamp_utc": "2026-04-11T20:16:00Z", "mode": "train", "global_step": 425, "epoch": 0.01641180105035527, "loss": 0.0023, "grad_norm": 8.99482536315918, "learning_rate": 8.715151515151515e-06, "num_tokens": 916259.0, "completions/mean_length": 45.625, "completions/min_length": 32.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.18696027994155884, "rewards/meter/std": 0.2777934968471527, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.18696027994155884, "rewards/total_composite/std": 0.2777934968471527, "reward": 0.18696027994155884, "reward_std": 0.2777934968471527, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08858486264944077, "sampling/sampling_logp_difference/max": 1.273571491241455, "sampling/importance_sampling_ratio/min": 0.2798304259777069, "sampling/importance_sampling_ratio/mean": 1.0027213096618652, "sampling/importance_sampling_ratio/max": 1.9791115522384644, "entropy": 0.5226194709539413, "clip_ratio/low_mean": 0.05649581435136497, "clip_ratio/low_min": 0.05649581435136497, "clip_ratio/high_mean": 0.025681341998279095, "clip_ratio/high_max": 0.025681341998279095, "clip_ratio/region_mean": 0.08217715634964406, "reward_total_mean": 0.18696027994155884, "reward_meter_mean": 0.18696027994155884, "reward_meter_std": 0.2777934968471527, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.18696027994155884, "reward_total_composite_std": 0.2777934968471527, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 425.0} {"timestamp_utc": "2026-04-11T20:16:05Z", "mode": "train", "global_step": 426, "epoch": 0.016450417052826693, "loss": 0.0217, "grad_norm": 5.837379455566406, "learning_rate": 8.712121212121212e-06, "num_tokens": 918078.0, "completions/mean_length": 71.375, "completions/min_length": 63.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.7622324228286743, "rewards/meter/std": 0.3091115951538086, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7051376700401306, "rewards/total_composite/std": 0.31919533014297485, "reward": 0.7051376700401306, "reward_std": 0.31919535994529724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07148178666830063, "sampling/sampling_logp_difference/max": 3.489603042602539, "sampling/importance_sampling_ratio/min": 0.030512982979416847, "sampling/importance_sampling_ratio/mean": 1.010563611984253, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.481033593416214, "clip_ratio/low_mean": 0.018849206971935928, "clip_ratio/low_min": 0.018849206971935928, "clip_ratio/high_mean": 0.04074844322167337, "clip_ratio/high_max": 0.04074844322167337, "clip_ratio/region_mean": 0.0595976501936093, "reward_total_mean": 0.7051376700401306, "reward_meter_mean": 0.7622324228286743, "reward_meter_std": 0.3091115951538086, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7051376700401306, "reward_total_composite_std": 0.31919533014297485, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 426.0} {"timestamp_utc": "2026-04-11T20:16:11Z", "mode": "train", "global_step": 427, "epoch": 0.016489033055298117, "loss": -0.0653, "grad_norm": 4.158021450042725, "learning_rate": 8.70909090909091e-06, "num_tokens": 920736.0, "completions/mean_length": 150.25, "completions/min_length": 129.0, "completions/max_length": 170.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.25, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.6729066967964172, "rewards/meter/std": 0.3364524841308594, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6173264980316162, "rewards/total_composite/std": 0.3513941466808319, "reward": 0.6173264980316162, "reward_std": 0.3513941466808319, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03079906478524208, "sampling/sampling_logp_difference/max": 2.866814613342285, "sampling/importance_sampling_ratio/min": 0.05687982589006424, "sampling/importance_sampling_ratio/mean": 0.9964794516563416, "sampling/importance_sampling_ratio/max": 1.959710717201233, "entropy": 0.14678781293332577, "clip_ratio/low_mean": 0.009528988506644964, "clip_ratio/low_min": 0.009528988506644964, "clip_ratio/high_mean": 0.023971143178641796, "clip_ratio/high_max": 0.023971143178641796, "clip_ratio/region_mean": 0.03350013168528676, "reward_total_mean": 0.6173264980316162, "reward_meter_mean": 0.6729066967964172, "reward_meter_std": 0.3364524841308594, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6173264980316162, "reward_total_composite_std": 0.3513941466808319, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 427.0} {"timestamp_utc": "2026-04-11T20:16:17Z", "mode": "train", "global_step": 428, "epoch": 0.01652764905776954, "loss": 0.0675, "grad_norm": 2.6976351737976074, "learning_rate": 8.706060606060607e-06, "num_tokens": 923000.0, "completions/mean_length": 129.0, "completions/min_length": 107.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.0, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.11112642288208008, "rewards/meter/std": 0.1817607432603836, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.11112642288208008, "rewards/total_composite/std": 0.1817607432603836, "reward": 0.11112642288208008, "reward_std": 0.1817607432603836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030291767790913582, "sampling/sampling_logp_difference/max": 2.559856414794922, "sampling/importance_sampling_ratio/min": 0.07731583714485168, "sampling/importance_sampling_ratio/mean": 1.0000661611557007, "sampling/importance_sampling_ratio/max": 1.9328269958496094, "entropy": 0.13649542536586523, "clip_ratio/low_mean": 0.009472908801399171, "clip_ratio/low_min": 0.009472908801399171, "clip_ratio/high_mean": 0.010739810299128294, "clip_ratio/high_max": 0.010739810299128294, "clip_ratio/region_mean": 0.020212719100527465, "reward_total_mean": 0.11112642288208008, "reward_meter_mean": 0.11112642288208008, "reward_meter_std": 0.1817607432603836, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.11112642288208008, "reward_total_composite_std": 0.1817607432603836, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 428.0} {"timestamp_utc": "2026-04-11T20:16:23Z", "mode": "train", "global_step": 429, "epoch": 0.016566265060240965, "loss": -0.0293, "grad_norm": 3.4749865531921387, "learning_rate": 8.703030303030304e-06, "num_tokens": 925409.0, "completions/mean_length": 127.125, "completions/min_length": 113.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.125, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.9445109963417053, "rewards/meter/std": 0.06675712764263153, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8244721293449402, "rewards/total_composite/std": 0.16515542566776276, "reward": 0.8244721293449402, "reward_std": 0.16515542566776276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05006590485572815, "sampling/sampling_logp_difference/max": 3.9263317584991455, "sampling/importance_sampling_ratio/min": 0.019715862348675728, "sampling/importance_sampling_ratio/mean": 0.9982576966285706, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26551887206733227, "clip_ratio/low_mean": 0.012388192000798881, "clip_ratio/low_min": 0.012388192000798881, "clip_ratio/high_mean": 0.02691519889049232, "clip_ratio/high_max": 0.02691519889049232, "clip_ratio/region_mean": 0.0393033908912912, "reward_total_mean": 0.8244721293449402, "reward_meter_mean": 0.9445109963417053, "reward_meter_std": 0.06675712764263153, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8244721293449402, "reward_total_composite_std": 0.16515542566776276, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 429.0} {"timestamp_utc": "2026-04-11T20:16:28Z", "mode": "train", "global_step": 430, "epoch": 0.01660488106271239, "loss": 0.0512, "grad_norm": 5.7448906898498535, "learning_rate": 8.700000000000001e-06, "num_tokens": 927497.0, "completions/mean_length": 98.0, "completions/min_length": 90.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.574688196182251, "rewards/meter/std": 0.4255768954753876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.574688196182251, "rewards/total_composite/std": 0.4255768954753876, "reward": 0.574688196182251, "reward_std": 0.42557692527770996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06919876486063004, "sampling/sampling_logp_difference/max": 1.1366691589355469, "sampling/importance_sampling_ratio/min": 0.3208860456943512, "sampling/importance_sampling_ratio/mean": 1.0072433948516846, "sampling/importance_sampling_ratio/max": 1.9886913299560547, "entropy": 0.5216472372412682, "clip_ratio/low_mean": 0.03095380635932088, "clip_ratio/low_min": 0.03095380635932088, "clip_ratio/high_mean": 0.03724813973531127, "clip_ratio/high_max": 0.03724813973531127, "clip_ratio/region_mean": 0.06820194609463215, "reward_total_mean": 0.574688196182251, "reward_meter_mean": 0.574688196182251, "reward_meter_std": 0.4255768954753876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.574688196182251, "reward_total_composite_std": 0.4255768954753876, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 430.0} {"timestamp_utc": "2026-04-11T20:16:33Z", "mode": "train", "global_step": 431, "epoch": 0.016643497065183813, "loss": -0.0277, "grad_norm": 14.200037956237793, "learning_rate": 8.696969696969699e-06, "num_tokens": 929098.0, "completions/mean_length": 34.125, "completions/min_length": 22.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.811887264251709, "rewards/meter/std": 0.2896690368652344, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.811887264251709, "rewards/total_composite/std": 0.2896690368652344, "reward": 0.811887264251709, "reward_std": 0.2896690368652344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09828025102615356, "sampling/sampling_logp_difference/max": 0.8195581436157227, "sampling/importance_sampling_ratio/min": 0.440626323223114, "sampling/importance_sampling_ratio/mean": 1.0314644575119019, "sampling/importance_sampling_ratio/max": 1.7877914905548096, "entropy": 0.9744241572916508, "clip_ratio/low_mean": 0.022997836116701365, "clip_ratio/low_min": 0.022997836116701365, "clip_ratio/high_mean": 0.060504904482513666, "clip_ratio/high_max": 0.060504904482513666, "clip_ratio/region_mean": 0.08350274059921503, "reward_total_mean": 0.811887264251709, "reward_meter_mean": 0.811887264251709, "reward_meter_std": 0.2896690368652344, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.811887264251709, "reward_total_composite_std": 0.2896690368652344, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 431.0} {"timestamp_utc": "2026-04-11T20:16:38Z", "mode": "train", "global_step": 432, "epoch": 0.016682113067655237, "loss": 0.038, "grad_norm": 6.835657596588135, "learning_rate": 8.693939393939394e-06, "num_tokens": 931202.0, "completions/mean_length": 103.0, "completions/min_length": 96.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.45791012048721313, "rewards/meter/std": 0.43955934047698975, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.45791012048721313, "rewards/total_composite/std": 0.43955934047698975, "reward": 0.45791012048721313, "reward_std": 0.43955934047698975, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025745589286088943, "sampling/sampling_logp_difference/max": 1.1592121124267578, "sampling/importance_sampling_ratio/min": 0.3137332797050476, "sampling/importance_sampling_ratio/mean": 0.9998974800109863, "sampling/importance_sampling_ratio/max": 1.681004524230957, "entropy": 0.12999565247446299, "clip_ratio/low_mean": 0.01565172686241567, "clip_ratio/low_min": 0.01565172686241567, "clip_ratio/high_mean": 0.0062806373462080956, "clip_ratio/high_max": 0.0062806373462080956, "clip_ratio/region_mean": 0.021932364208623767, "reward_total_mean": 0.45791012048721313, "reward_meter_mean": 0.45791012048721313, "reward_meter_std": 0.43955934047698975, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.45791012048721313, "reward_total_composite_std": 0.43955934047698975, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 432.0} {"timestamp_utc": "2026-04-11T20:16:44Z", "mode": "train", "global_step": 433, "epoch": 0.01672072907012666, "loss": 0.0363, "grad_norm": 9.083555221557617, "learning_rate": 8.690909090909091e-06, "num_tokens": 933225.0, "completions/mean_length": 94.875, "completions/min_length": 65.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.8027656674385071, "rewards/meter/std": 0.3288112282752991, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7613606452941895, "rewards/total_composite/std": 0.3221178352832794, "reward": 0.7613606452941895, "reward_std": 0.3221178352832794, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08027959614992142, "sampling/sampling_logp_difference/max": 1.461640477180481, "sampling/importance_sampling_ratio/min": 0.2318556159734726, "sampling/importance_sampling_ratio/mean": 1.0011898279190063, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45340409502387047, "clip_ratio/low_mean": 0.0251602572388947, "clip_ratio/low_min": 0.0251602572388947, "clip_ratio/high_mean": 0.03764841705560684, "clip_ratio/high_max": 0.03764841705560684, "clip_ratio/region_mean": 0.06280867429450154, "reward_total_mean": 0.7613606452941895, "reward_meter_mean": 0.8027656674385071, "reward_meter_std": 0.3288112282752991, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7613606452941895, "reward_total_composite_std": 0.3221178352832794, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 433.0} {"timestamp_utc": "2026-04-11T20:16:49Z", "mode": "train", "global_step": 434, "epoch": 0.016759345072598086, "loss": 0.007, "grad_norm": 3.4593441486358643, "learning_rate": 8.687878787878789e-06, "num_tokens": 935461.0, "completions/mean_length": 114.5, "completions/min_length": 104.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.5, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.8310332298278809, "rewards/meter/std": 0.32082754373550415, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8310332298278809, "rewards/total_composite/std": 0.32082754373550415, "reward": 0.8310332298278809, "reward_std": 0.32082754373550415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032745786011219025, "sampling/sampling_logp_difference/max": 6.637192249298096, "sampling/importance_sampling_ratio/min": 0.0013107022969052196, "sampling/importance_sampling_ratio/mean": 1.0050222873687744, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11784437298774719, "clip_ratio/low_mean": 0.0010869564721360803, "clip_ratio/low_min": 0.0010869564721360803, "clip_ratio/high_mean": 0.013409517356194556, "clip_ratio/high_max": 0.013409517356194556, "clip_ratio/region_mean": 0.014496473828330636, "reward_total_mean": 0.8310332298278809, "reward_meter_mean": 0.8310332298278809, "reward_meter_std": 0.32082754373550415, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8310332298278809, "reward_total_composite_std": 0.32082754373550415, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 434.0} {"timestamp_utc": "2026-04-11T20:16:55Z", "mode": "train", "global_step": 435, "epoch": 0.01679796107506951, "loss": -0.0446, "grad_norm": 5.847456455230713, "learning_rate": 8.684848484848486e-06, "num_tokens": 937614.0, "completions/mean_length": 101.125, "completions/min_length": 68.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.8531937599182129, "rewards/meter/std": 0.27812930941581726, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8126308917999268, "rewards/total_composite/std": 0.2817155122756958, "reward": 0.8126308917999268, "reward_std": 0.2817155420780182, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07007157802581787, "sampling/sampling_logp_difference/max": 2.1144661903381348, "sampling/importance_sampling_ratio/min": 0.12069769948720932, "sampling/importance_sampling_ratio/mean": 1.002807378768921, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5314316283911467, "clip_ratio/low_mean": 0.014399510342627764, "clip_ratio/low_min": 0.014399510342627764, "clip_ratio/high_mean": 0.04601668380200863, "clip_ratio/high_max": 0.04601668380200863, "clip_ratio/region_mean": 0.06041619414463639, "reward_total_mean": 0.8126308917999268, "reward_meter_mean": 0.8531937599182129, "reward_meter_std": 0.27812930941581726, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8126308917999268, "reward_total_composite_std": 0.2817155122756958, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 435.0} {"timestamp_utc": "2026-04-11T20:17:00Z", "mode": "train", "global_step": 436, "epoch": 0.016836577077540934, "loss": -0.0109, "grad_norm": 5.732955455780029, "learning_rate": 8.681818181818182e-06, "num_tokens": 939392.0, "completions/mean_length": 77.25, "completions/min_length": 72.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.25, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.3143185079097748, "rewards/meter/std": 0.41200879216194153, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3143185079097748, "rewards/total_composite/std": 0.41200879216194153, "reward": 0.3143185079097748, "reward_std": 0.41200879216194153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04445616900920868, "sampling/sampling_logp_difference/max": 4.05485725402832, "sampling/importance_sampling_ratio/min": 0.01733795367181301, "sampling/importance_sampling_ratio/mean": 1.0097142457962036, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1496438980102539, "clip_ratio/low_mean": 0.026244841050356627, "clip_ratio/low_min": 0.026244841050356627, "clip_ratio/high_mean": 0.0015625000232830644, "clip_ratio/high_max": 0.0015625000232830644, "clip_ratio/region_mean": 0.02780734107363969, "reward_total_mean": 0.3143185079097748, "reward_meter_mean": 0.3143185079097748, "reward_meter_std": 0.41200879216194153, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3143185079097748, "reward_total_composite_std": 0.41200879216194153, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 436.0} {"timestamp_utc": "2026-04-11T20:17:05Z", "mode": "train", "global_step": 437, "epoch": 0.016875193080012358, "loss": -0.0275, "grad_norm": 4.509356498718262, "learning_rate": 8.67878787878788e-06, "num_tokens": 941350.0, "completions/mean_length": 82.75, "completions/min_length": 43.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.2568199038505554, "rewards/meter/std": 0.359527051448822, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.2339237928390503, "rewards/total_composite/std": 0.3625546097755432, "reward": 0.2339237928390503, "reward_std": 0.3625546097755432, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.058143459260463715, "sampling/sampling_logp_difference/max": 1.3640966415405273, "sampling/importance_sampling_ratio/min": 0.25561147928237915, "sampling/importance_sampling_ratio/mean": 1.0075230598449707, "sampling/importance_sampling_ratio/max": 1.8314160108566284, "entropy": 0.3760291952639818, "clip_ratio/low_mean": 0.01840795623138547, "clip_ratio/low_min": 0.01840795623138547, "clip_ratio/high_mean": 0.014880952425301075, "clip_ratio/high_max": 0.014880952425301075, "clip_ratio/region_mean": 0.033288908656686544, "reward_total_mean": 0.2339237928390503, "reward_meter_mean": 0.2568199038505554, "reward_meter_std": 0.359527051448822, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.2339237928390503, "reward_total_composite_std": 0.3625546097755432, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 437.0} {"timestamp_utc": "2026-04-11T20:17:15Z", "mode": "train", "global_step": 438, "epoch": 0.016913809082483782, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 8.675757575757576e-06, "num_tokens": 943046.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9948071241378784, "rewards/meter/std": 0.00352023309096694, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7461053729057312, "rewards/total_composite/std": 0.0026401823852211237, "reward": 0.7461053729057312, "reward_std": 0.0026401823852211237, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7461053729057312, "reward_meter_mean": 0.9948071241378784, "reward_meter_std": 0.00352023309096694, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7461053729057312, "reward_total_composite_std": 0.0026401823852211237, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 438.0} {"timestamp_utc": "2026-04-11T20:17:21Z", "mode": "train", "global_step": 439, "epoch": 0.016952425084955206, "loss": 0.0153, "grad_norm": 3.7757630348205566, "learning_rate": 8.672727272727273e-06, "num_tokens": 945072.0, "completions/mean_length": 94.25, "completions/min_length": 79.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.25, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.8440791368484497, "rewards/meter/std": 0.29224100708961487, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7195857763290405, "rewards/total_composite/std": 0.4076501429080963, "reward": 0.7195857763290405, "reward_std": 0.4076501131057739, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031045421957969666, "sampling/sampling_logp_difference/max": 1.8811311721801758, "sampling/importance_sampling_ratio/min": 0.1524176001548767, "sampling/importance_sampling_ratio/mean": 1.0024596452713013, "sampling/importance_sampling_ratio/max": 1.6128379106521606, "entropy": 0.20515940617769957, "clip_ratio/low_mean": 0.006527067394927144, "clip_ratio/low_min": 0.006527067394927144, "clip_ratio/high_mean": 0.023564013885334134, "clip_ratio/high_max": 0.023564013885334134, "clip_ratio/region_mean": 0.030091081280261278, "reward_total_mean": 0.7195857763290405, "reward_meter_mean": 0.8440791368484497, "reward_meter_std": 0.29224100708961487, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7195857763290405, "reward_total_composite_std": 0.4076501429080963, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 439.0} {"timestamp_utc": "2026-04-11T20:17:28Z", "mode": "train", "global_step": 440, "epoch": 0.01699104108742663, "loss": -0.0297, "grad_norm": 4.999176979064941, "learning_rate": 8.66969696969697e-06, "num_tokens": 948306.0, "completions/mean_length": 199.25, "completions/min_length": 179.0, "completions/max_length": 223.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 199.25, "completions/min_terminated_length": 179.0, "completions/max_terminated_length": 223.0, "rewards/meter/mean": 0.906821608543396, "rewards/meter/std": 0.20782814919948578, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.1035098284482956, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.800205647945404, "rewards/total_composite/std": 0.22486315667629242, "reward": 0.800205647945404, "reward_std": 0.22486314177513123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024938292801380157, "sampling/sampling_logp_difference/max": 2.0006561279296875, "sampling/importance_sampling_ratio/min": 0.13524653017520905, "sampling/importance_sampling_ratio/mean": 1.0032387971878052, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1384673686698079, "clip_ratio/low_mean": 0.013955606264062226, "clip_ratio/low_min": 0.013955606264062226, "clip_ratio/high_mean": 0.008545128046534956, "clip_ratio/high_max": 0.008545128046534956, "clip_ratio/region_mean": 0.02250073431059718, "reward_total_mean": 0.800205647945404, "reward_meter_mean": 0.906821608543396, "reward_meter_std": 0.20782814919948578, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.1035098284482956, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.800205647945404, "reward_total_composite_std": 0.22486315667629242, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 440.0} {"timestamp_utc": "2026-04-11T20:17:32Z", "mode": "train", "global_step": 441, "epoch": 0.017029657089898054, "loss": 0.043, "grad_norm": 12.705753326416016, "learning_rate": 8.666666666666668e-06, "num_tokens": 949895.0, "completions/mean_length": 33.625, "completions/min_length": 27.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.7382475137710571, "rewards/meter/std": 0.344305157661438, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7382475137710571, "rewards/total_composite/std": 0.344305157661438, "reward": 0.7382475137710571, "reward_std": 0.344305157661438, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11312177032232285, "sampling/sampling_logp_difference/max": 1.377732753753662, "sampling/importance_sampling_ratio/min": 0.2521496117115021, "sampling/importance_sampling_ratio/mean": 1.0123142004013062, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6665975973010063, "clip_ratio/low_mean": 0.04578754771500826, "clip_ratio/low_min": 0.04578754771500826, "clip_ratio/high_mean": 0.06254499591886997, "clip_ratio/high_max": 0.06254499591886997, "clip_ratio/region_mean": 0.10833254363387823, "reward_total_mean": 0.7382475137710571, "reward_meter_mean": 0.7382475137710571, "reward_meter_std": 0.344305157661438, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7382475137710571, "reward_total_composite_std": 0.344305157661438, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 441.0} {"timestamp_utc": "2026-04-11T20:17:37Z", "mode": "train", "global_step": 442, "epoch": 0.01706827309236948, "loss": 0.0748, "grad_norm": 6.052305698394775, "learning_rate": 8.663636363636363e-06, "num_tokens": 951634.0, "completions/mean_length": 67.375, "completions/min_length": 59.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.375, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.6851848363876343, "rewards/meter/std": 0.40763622522354126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6851848363876343, "rewards/total_composite/std": 0.40763622522354126, "reward": 0.6851848363876343, "reward_std": 0.40763622522354126, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06014850363135338, "sampling/sampling_logp_difference/max": 1.9354361295700073, "sampling/importance_sampling_ratio/min": 0.14436128735542297, "sampling/importance_sampling_ratio/mean": 1.01041579246521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44354628771543503, "clip_ratio/low_mean": 0.018479025457054377, "clip_ratio/low_min": 0.018479025457054377, "clip_ratio/high_mean": 0.04311846289783716, "clip_ratio/high_max": 0.04311846289783716, "clip_ratio/region_mean": 0.06159748835489154, "reward_total_mean": 0.6851848363876343, "reward_meter_mean": 0.6851848363876343, "reward_meter_std": 0.40763622522354126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6851848363876343, "reward_total_composite_std": 0.40763622522354126, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 442.0} {"timestamp_utc": "2026-04-11T20:17:42Z", "mode": "train", "global_step": 443, "epoch": 0.017106889094840903, "loss": 0.0122, "grad_norm": 9.460426330566406, "learning_rate": 8.660606060606062e-06, "num_tokens": 953439.0, "completions/mean_length": 68.625, "completions/min_length": 57.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.625, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.861977219581604, "rewards/meter/std": 0.24617436528205872, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.861977219581604, "rewards/total_composite/std": 0.24617436528205872, "reward": 0.861977219581604, "reward_std": 0.24617433547973633, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06774439662694931, "sampling/sampling_logp_difference/max": 1.9731206893920898, "sampling/importance_sampling_ratio/min": 0.1390223354101181, "sampling/importance_sampling_ratio/mean": 1.0046181678771973, "sampling/importance_sampling_ratio/max": 1.7051851749420166, "entropy": 0.5356311798095703, "clip_ratio/low_mean": 0.014395925216376781, "clip_ratio/low_min": 0.014395925216376781, "clip_ratio/high_mean": 0.04367695190012455, "clip_ratio/high_max": 0.04367695190012455, "clip_ratio/region_mean": 0.05807287711650133, "reward_total_mean": 0.861977219581604, "reward_meter_mean": 0.861977219581604, "reward_meter_std": 0.24617436528205872, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.861977219581604, "reward_total_composite_std": 0.24617436528205872, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 443.0} {"timestamp_utc": "2026-04-11T20:17:48Z", "mode": "train", "global_step": 444, "epoch": 0.017145505097312327, "loss": 0.0611, "grad_norm": 6.638420104980469, "learning_rate": 8.657575757575758e-06, "num_tokens": 955748.0, "completions/mean_length": 106.625, "completions/min_length": 95.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.625, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.24504423141479492, "rewards/meter/std": 0.346823513507843, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.24504423141479492, "rewards/total_composite/std": 0.346823513507843, "reward": 0.24504423141479492, "reward_std": 0.346823513507843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05710998922586441, "sampling/sampling_logp_difference/max": 1.4080891609191895, "sampling/importance_sampling_ratio/min": 0.2446102499961853, "sampling/importance_sampling_ratio/mean": 1.0113799571990967, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31410662829875946, "clip_ratio/low_mean": 0.05329327145591378, "clip_ratio/low_min": 0.05329327145591378, "clip_ratio/high_mean": 0.011721267364919186, "clip_ratio/high_max": 0.011721267364919186, "clip_ratio/region_mean": 0.06501453882083297, "reward_total_mean": 0.24504423141479492, "reward_meter_mean": 0.24504423141479492, "reward_meter_std": 0.346823513507843, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.24504423141479492, "reward_total_composite_std": 0.346823513507843, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 444.0} {"timestamp_utc": "2026-04-11T20:17:53Z", "mode": "train", "global_step": 445, "epoch": 0.01718412109978375, "loss": 0.0332, "grad_norm": 9.578984260559082, "learning_rate": 8.654545454545455e-06, "num_tokens": 957485.0, "completions/mean_length": 68.125, "completions/min_length": 62.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.7745425701141357, "rewards/meter/std": 0.3829474449157715, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7745425701141357, "rewards/total_composite/std": 0.3829474449157715, "reward": 0.7745425701141357, "reward_std": 0.38294747471809387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08342333137989044, "sampling/sampling_logp_difference/max": 1.1612434387207031, "sampling/importance_sampling_ratio/min": 0.31309664249420166, "sampling/importance_sampling_ratio/mean": 1.0089476108551025, "sampling/importance_sampling_ratio/max": 1.9029569625854492, "entropy": 0.7675438597798347, "clip_ratio/low_mean": 0.021353119518607855, "clip_ratio/low_min": 0.021353119518607855, "clip_ratio/high_mean": 0.05313561297953129, "clip_ratio/high_max": 0.05313561297953129, "clip_ratio/region_mean": 0.07448873249813914, "reward_total_mean": 0.7745425701141357, "reward_meter_mean": 0.7745425701141357, "reward_meter_std": 0.3829474449157715, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7745425701141357, "reward_total_composite_std": 0.3829474449157715, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 445.0} {"timestamp_utc": "2026-04-11T20:17:57Z", "mode": "train", "global_step": 446, "epoch": 0.017222737102255175, "loss": 0.0135, "grad_norm": 16.04683494567871, "learning_rate": 8.651515151515152e-06, "num_tokens": 959040.0, "completions/mean_length": 38.375, "completions/min_length": 32.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8562139272689819, "rewards/meter/std": 0.34138572216033936, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8562139272689819, "rewards/total_composite/std": 0.34138572216033936, "reward": 0.8562139272689819, "reward_std": 0.34138569235801697, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09310141205787659, "sampling/sampling_logp_difference/max": 1.2954916954040527, "sampling/importance_sampling_ratio/min": 0.2737632095813751, "sampling/importance_sampling_ratio/mean": 1.0087430477142334, "sampling/importance_sampling_ratio/max": 1.844504952430725, "entropy": 0.6561026684939861, "clip_ratio/low_mean": 0.02302631549537182, "clip_ratio/low_min": 0.02302631549537182, "clip_ratio/high_mean": 0.05887592723593116, "clip_ratio/high_max": 0.05887592723593116, "clip_ratio/region_mean": 0.08190224273130298, "reward_total_mean": 0.8562139272689819, "reward_meter_mean": 0.8562139272689819, "reward_meter_std": 0.34138572216033936, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8562139272689819, "reward_total_composite_std": 0.34138572216033936, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 446.0} {"timestamp_utc": "2026-04-11T20:18:07Z", "mode": "train", "global_step": 447, "epoch": 0.0172613531047266, "loss": -0.2555, "grad_norm": 0.7952864766120911, "learning_rate": 8.64848484848485e-06, "num_tokens": 961699.0, "completions/mean_length": 230.375, "completions/min_length": 177.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 190.1428680419922, "completions/min_terminated_length": 177.0, "completions/max_terminated_length": 197.0, "rewards/meter/mean": 0.9410502910614014, "rewards/meter/std": 0.1454276442527771, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8865582346916199, "rewards/total_composite/std": 0.29953092336654663, "reward": 0.8865582346916199, "reward_std": 0.29953086376190186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01994623802602291, "sampling/sampling_logp_difference/max": 1.1244478225708008, "sampling/importance_sampling_ratio/min": 0.3248317837715149, "sampling/importance_sampling_ratio/mean": 1.0042375326156616, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12354243081063032, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.011402326985262334, "clip_ratio/high_max": 0.011402326985262334, "clip_ratio/region_mean": 0.011402326985262334, "reward_total_mean": 0.8865582346916199, "reward_meter_mean": 0.9410502910614014, "reward_meter_std": 0.1454276442527771, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8865582346916199, "reward_total_composite_std": 0.29953092336654663, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 447.0} {"timestamp_utc": "2026-04-11T20:18:13Z", "mode": "train", "global_step": 448, "epoch": 0.017299969107198023, "loss": 0.0638, "grad_norm": 2.0035412311553955, "learning_rate": 8.645454545454545e-06, "num_tokens": 964297.0, "completions/mean_length": 158.75, "completions/min_length": 126.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.75, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.859869122505188, "rewards/meter/std": 0.3417814075946808, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.859869122505188, "rewards/total_composite/std": 0.3417814075946808, "reward": 0.859869122505188, "reward_std": 0.3417814075946808, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03218178451061249, "sampling/sampling_logp_difference/max": 5.192971706390381, "sampling/importance_sampling_ratio/min": 0.005555473268032074, "sampling/importance_sampling_ratio/mean": 0.9988955855369568, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10146280331537127, "clip_ratio/low_mean": 0.002027027076110244, "clip_ratio/low_min": 0.002027027076110244, "clip_ratio/high_mean": 0.01393397233914584, "clip_ratio/high_max": 0.01393397233914584, "clip_ratio/region_mean": 0.015960999415256083, "reward_total_mean": 0.859869122505188, "reward_meter_mean": 0.859869122505188, "reward_meter_std": 0.3417814075946808, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.859869122505188, "reward_total_composite_std": 0.3417814075946808, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 448.0} {"timestamp_utc": "2026-04-11T20:18:19Z", "mode": "train", "global_step": 449, "epoch": 0.017338585109669447, "loss": 0.0171, "grad_norm": 6.3059892654418945, "learning_rate": 8.642424242424242e-06, "num_tokens": 966255.0, "completions/mean_length": 82.75, "completions/min_length": 78.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.75, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.6425033807754517, "rewards/meter/std": 0.41706737875938416, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6425033807754517, "rewards/total_composite/std": 0.41706737875938416, "reward": 0.6425033807754517, "reward_std": 0.41706737875938416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.053077779710292816, "sampling/sampling_logp_difference/max": 3.8796114921569824, "sampling/importance_sampling_ratio/min": 0.02065885066986084, "sampling/importance_sampling_ratio/mean": 1.0040130615234375, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2209324724972248, "clip_ratio/low_mean": 0.007736280560493469, "clip_ratio/low_min": 0.007736280560493469, "clip_ratio/high_mean": 0.029356154147535563, "clip_ratio/high_max": 0.029356154147535563, "clip_ratio/region_mean": 0.03709243470802903, "reward_total_mean": 0.6425033807754517, "reward_meter_mean": 0.6425033807754517, "reward_meter_std": 0.41706737875938416, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6425033807754517, "reward_total_composite_std": 0.41706737875938416, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 449.0} {"timestamp_utc": "2026-04-11T20:18:23Z", "mode": "train", "global_step": 450, "epoch": 0.01737720111214087, "loss": 0.0266, "grad_norm": 6.594456195831299, "learning_rate": 8.63939393939394e-06, "num_tokens": 968263.0, "completions/mean_length": 81.0, "completions/min_length": 73.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.861186146736145, "rewards/meter/std": 0.25013267993927, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.861186146736145, "rewards/total_composite/std": 0.25013267993927, "reward": 0.861186146736145, "reward_std": 0.25013265013694763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030779149383306503, "sampling/sampling_logp_difference/max": 1.7460429668426514, "sampling/importance_sampling_ratio/min": 0.1744629293680191, "sampling/importance_sampling_ratio/mean": 0.9992127418518066, "sampling/importance_sampling_ratio/max": 1.9482147693634033, "entropy": 0.15494854096323252, "clip_ratio/low_mean": 0.004411764908581972, "clip_ratio/low_min": 0.004411764908581972, "clip_ratio/high_mean": 0.027218999108299613, "clip_ratio/high_max": 0.027218999108299613, "clip_ratio/region_mean": 0.031630764016881585, "reward_total_mean": 0.861186146736145, "reward_meter_mean": 0.861186146736145, "reward_meter_std": 0.25013267993927, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.861186146736145, "reward_total_composite_std": 0.25013267993927, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 450.0} {"timestamp_utc": "2026-04-11T20:19:54Z", "mode": "eval", "global_step": 450, "epoch": 0.01737720111214087, "eval_loss": NaN, "eval_runtime": 90.6979, "eval_samples_per_second": 1.147, "eval_steps_per_second": 0.143, "eval_num_tokens": 968263.0, "eval_completions/mean_length": 252.39423076923077, "eval_completions/min_length": 61.23076923076923, "eval_completions/max_length": 485.38461538461536, "eval_completions/clipped_ratio": 0.10576923076923077, "eval_completions/mean_terminated_length": 221.38736900916467, "eval_completions/min_terminated_length": 61.23076923076923, "eval_completions/max_terminated_length": 421.0769230769231, "eval_rewards/meter/mean": 0.7144990059045645, "eval_rewards/meter/std": 0.3781636357307434, "eval_rewards/count_adherence/mean": 0.9385907145646902, "eval_rewards/count_adherence/std": 0.09539258336791626, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/total_composite/mean": 0.6704203371818249, "eval_rewards/total_composite/std": 0.37348280388575333, "eval_reward": 0.6704203371818249, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.011250602045597939, "eval_sampling/sampling_logp_difference/max": 0.8244214149621817, "eval_sampling/importance_sampling_ratio/min": 0.45446773446523225, "eval_sampling/importance_sampling_ratio/mean": 1.002858510384193, "eval_sampling/importance_sampling_ratio/max": 1.352581189228938, "eval_entropy": 0.11755917536524627, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6704203371818249, "eval_reward_meter_mean": 0.7144990059045645, "eval_reward_meter_std": 0.3781636357307434, "eval_reward_count_adherence_mean": 0.9385907145646902, "eval_reward_count_adherence_std": 0.09539258336791626, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_total_composite_mean": 0.6704203371818249, "eval_reward_total_composite_std": 0.37348280388575333, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 450.0} {"timestamp_utc": "2026-04-11T20:20:01Z", "mode": "train", "global_step": 451, "epoch": 0.017415817114612295, "loss": -0.0468, "grad_norm": 8.833879470825195, "learning_rate": 8.636363636363637e-06, "num_tokens": 969993.0, "completions/mean_length": 60.25, "completions/min_length": 48.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.6889505386352539, "rewards/meter/std": 0.3999699652194977, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6889505386352539, "rewards/total_composite/std": 0.3999699652194977, "reward": 0.6889505386352539, "reward_std": 0.3999699354171753, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09507620334625244, "sampling/sampling_logp_difference/max": 1.3110675811767578, "sampling/importance_sampling_ratio/min": 0.35178372263908386, "sampling/importance_sampling_ratio/mean": 1.0142425298690796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7900894470512867, "clip_ratio/low_mean": 0.03685897495597601, "clip_ratio/low_min": 0.03685897495597601, "clip_ratio/high_mean": 0.0517552737146616, "clip_ratio/high_max": 0.0517552737146616, "clip_ratio/region_mean": 0.08861424867063761, "reward_total_mean": 0.6889505386352539, "reward_meter_mean": 0.6889505386352539, "reward_meter_std": 0.3999699652194977, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6889505386352539, "reward_total_composite_std": 0.3999699652194977, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 451.0} {"timestamp_utc": "2026-04-11T20:20:07Z", "mode": "train", "global_step": 452, "epoch": 0.01745443311708372, "loss": -0.0294, "grad_norm": 4.048279762268066, "learning_rate": 8.633333333333334e-06, "num_tokens": 972209.0, "completions/mean_length": 94.0, "completions/min_length": 87.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.0, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9136518239974976, "rewards/meter/std": 0.0902651846408844, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9136518239974976, "rewards/total_composite/std": 0.0902651846408844, "reward": 0.9136518239974976, "reward_std": 0.0902651846408844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03461015596985817, "sampling/sampling_logp_difference/max": 1.1600208282470703, "sampling/importance_sampling_ratio/min": 0.3134796619415283, "sampling/importance_sampling_ratio/mean": 1.0003708600997925, "sampling/importance_sampling_ratio/max": 1.876058578491211, "entropy": 0.20060661621391773, "clip_ratio/low_mean": 0.006992337410338223, "clip_ratio/low_min": 0.006992337410338223, "clip_ratio/high_mean": 0.02525400766171515, "clip_ratio/high_max": 0.02525400766171515, "clip_ratio/region_mean": 0.03224634507205337, "reward_total_mean": 0.9136518239974976, "reward_meter_mean": 0.9136518239974976, "reward_meter_std": 0.0902651846408844, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9136518239974976, "reward_total_composite_std": 0.0902651846408844, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 452.0} {"timestamp_utc": "2026-04-11T20:20:12Z", "mode": "train", "global_step": 453, "epoch": 0.017493049119555144, "loss": 0.0093, "grad_norm": 6.656115531921387, "learning_rate": 8.630303030303032e-06, "num_tokens": 973962.0, "completions/mean_length": 70.125, "completions/min_length": 68.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9888208508491516, "rewards/meter/std": 0.006749412976205349, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9888208508491516, "rewards/total_composite/std": 0.006749412976205349, "reward": 0.9888208508491516, "reward_std": 0.006749419495463371, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02052040584385395, "sampling/sampling_logp_difference/max": 1.512446403503418, "sampling/importance_sampling_ratio/min": 0.22037020325660706, "sampling/importance_sampling_ratio/mean": 0.9998013377189636, "sampling/importance_sampling_ratio/max": 1.4743931293487549, "entropy": 0.08665410289540887, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.015870221075601876, "clip_ratio/high_max": 0.015870221075601876, "clip_ratio/region_mean": 0.015870221075601876, "reward_total_mean": 0.9888208508491516, "reward_meter_mean": 0.9888208508491516, "reward_meter_std": 0.006749412976205349, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9888208508491516, "reward_total_composite_std": 0.006749412976205349, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 453.0} {"timestamp_utc": "2026-04-11T20:20:17Z", "mode": "train", "global_step": 454, "epoch": 0.017531665122026568, "loss": -0.0522, "grad_norm": 9.838415145874023, "learning_rate": 8.627272727272727e-06, "num_tokens": 975703.0, "completions/mean_length": 58.625, "completions/min_length": 45.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.625, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9708679914474487, "rewards/meter/std": 0.03722724691033363, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9708679914474487, "rewards/total_composite/std": 0.03722724691033363, "reward": 0.9708679914474487, "reward_std": 0.037227239459753036, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07885369658470154, "sampling/sampling_logp_difference/max": 1.6653971672058105, "sampling/importance_sampling_ratio/min": 0.18911553919315338, "sampling/importance_sampling_ratio/mean": 1.0130358934402466, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6852592751383781, "clip_ratio/low_mean": 0.01666666753590107, "clip_ratio/low_min": 0.01666666753590107, "clip_ratio/high_mean": 0.0574263078160584, "clip_ratio/high_max": 0.0574263078160584, "clip_ratio/region_mean": 0.07409297535195947, "reward_total_mean": 0.9708679914474487, "reward_meter_mean": 0.9708679914474487, "reward_meter_std": 0.03722724691033363, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9708679914474487, "reward_total_composite_std": 0.03722724691033363, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 454.0} {"timestamp_utc": "2026-04-11T20:20:22Z", "mode": "train", "global_step": 455, "epoch": 0.017570281124497992, "loss": 0.0665, "grad_norm": 7.684765815734863, "learning_rate": 8.624242424242424e-06, "num_tokens": 977683.0, "completions/mean_length": 62.5, "completions/min_length": 48.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.6929647922515869, "rewards/meter/std": 0.3059764504432678, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6929647922515869, "rewards/total_composite/std": 0.3059764504432678, "reward": 0.6929647922515869, "reward_std": 0.3059764504432678, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07383247464895248, "sampling/sampling_logp_difference/max": 2.803025007247925, "sampling/importance_sampling_ratio/min": 0.06062639132142067, "sampling/importance_sampling_ratio/mean": 1.0214698314666748, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5540340095758438, "clip_ratio/low_mean": 0.026715174899436533, "clip_ratio/low_min": 0.026715174899436533, "clip_ratio/high_mean": 0.03188905236311257, "clip_ratio/high_max": 0.03188905236311257, "clip_ratio/region_mean": 0.0586042272625491, "reward_total_mean": 0.6929647922515869, "reward_meter_mean": 0.6929647922515869, "reward_meter_std": 0.3059764504432678, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6929647922515869, "reward_total_composite_std": 0.3059764504432678, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 455.0} {"timestamp_utc": "2026-04-11T20:20:27Z", "mode": "train", "global_step": 456, "epoch": 0.017608897126969416, "loss": 0.0137, "grad_norm": 7.524571895599365, "learning_rate": 8.621212121212122e-06, "num_tokens": 979768.0, "completions/mean_length": 75.625, "completions/min_length": 71.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.625, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.8802084922790527, "rewards/meter/std": 0.23987267911434174, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8802084922790527, "rewards/total_composite/std": 0.23987267911434174, "reward": 0.8802084922790527, "reward_std": 0.23987266421318054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0491391159594059, "sampling/sampling_logp_difference/max": 2.2689783573150635, "sampling/importance_sampling_ratio/min": 0.1034177839756012, "sampling/importance_sampling_ratio/mean": 1.0096123218536377, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3263458888977766, "clip_ratio/low_mean": 0.004999999888241291, "clip_ratio/low_min": 0.004999999888241291, "clip_ratio/high_mean": 0.037544333608821034, "clip_ratio/high_max": 0.037544333608821034, "clip_ratio/region_mean": 0.042544333497062325, "reward_total_mean": 0.8802084922790527, "reward_meter_mean": 0.8802084922790527, "reward_meter_std": 0.23987267911434174, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8802084922790527, "reward_total_composite_std": 0.23987267911434174, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 456.0} {"timestamp_utc": "2026-04-11T20:20:33Z", "mode": "train", "global_step": 457, "epoch": 0.01764751312944084, "loss": 0.0947, "grad_norm": 6.170909404754639, "learning_rate": 8.618181818181819e-06, "num_tokens": 982005.0, "completions/mean_length": 110.625, "completions/min_length": 93.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.625, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.37684983015060425, "rewards/meter/std": 0.48171141743659973, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.3757762312889099, "rewards/total_composite/std": 0.4826580584049225, "reward": 0.3757762312889099, "reward_std": 0.4826580584049225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0551149956882, "sampling/sampling_logp_difference/max": 1.4722728729248047, "sampling/importance_sampling_ratio/min": 0.22940349578857422, "sampling/importance_sampling_ratio/mean": 1.0049611330032349, "sampling/importance_sampling_ratio/max": 1.8442530632019043, "entropy": 0.3454656656831503, "clip_ratio/low_mean": 0.026498167659156024, "clip_ratio/low_min": 0.026498167659156024, "clip_ratio/high_mean": 0.018728531897068024, "clip_ratio/high_max": 0.018728531897068024, "clip_ratio/region_mean": 0.04522669955622405, "reward_total_mean": 0.3757762312889099, "reward_meter_mean": 0.37684983015060425, "reward_meter_std": 0.48171141743659973, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.3757762312889099, "reward_total_composite_std": 0.4826580584049225, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 457.0} {"timestamp_utc": "2026-04-11T20:20:37Z", "mode": "train", "global_step": 458, "epoch": 0.017686129131912264, "loss": -0.0317, "grad_norm": 16.211706161499023, "learning_rate": 8.615151515151516e-06, "num_tokens": 983398.0, "completions/mean_length": 25.125, "completions/min_length": 10.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.125, "completions/min_terminated_length": 10.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.6279522180557251, "rewards/meter/std": 0.4916090965270996, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6279522180557251, "rewards/total_composite/std": 0.4916090965270996, "reward": 0.6279522180557251, "reward_std": 0.491609126329422, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1369742900133133, "sampling/sampling_logp_difference/max": 1.209054946899414, "sampling/importance_sampling_ratio/min": 0.29847922921180725, "sampling/importance_sampling_ratio/mean": 1.0198932886123657, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.231294609606266, "clip_ratio/low_mean": 0.04874078743159771, "clip_ratio/low_min": 0.04874078743159771, "clip_ratio/high_mean": 0.07460826355963945, "clip_ratio/high_max": 0.07460826355963945, "clip_ratio/region_mean": 0.12334905099123716, "reward_total_mean": 0.6279522180557251, "reward_meter_mean": 0.6279522180557251, "reward_meter_std": 0.4916090965270996, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6279522180557251, "reward_total_composite_std": 0.4916090965270996, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 458.0} {"timestamp_utc": "2026-04-11T20:20:42Z", "mode": "train", "global_step": 459, "epoch": 0.01772474513438369, "loss": 0.1012, "grad_norm": 7.541397571563721, "learning_rate": 8.612121212121213e-06, "num_tokens": 985277.0, "completions/mean_length": 74.875, "completions/min_length": 49.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.875, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.500255286693573, "rewards/meter/std": 0.3483890891075134, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.500255286693573, "rewards/total_composite/std": 0.3483890891075134, "reward": 0.500255286693573, "reward_std": 0.3483890891075134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08134466409683228, "sampling/sampling_logp_difference/max": 1.2600550651550293, "sampling/importance_sampling_ratio/min": 0.2836383879184723, "sampling/importance_sampling_ratio/mean": 1.000501036643982, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4475377518683672, "clip_ratio/low_mean": 0.02632259437814355, "clip_ratio/low_min": 0.02632259437814355, "clip_ratio/high_mean": 0.04640375077724457, "clip_ratio/high_max": 0.04640375077724457, "clip_ratio/region_mean": 0.07272634515538812, "reward_total_mean": 0.500255286693573, "reward_meter_mean": 0.500255286693573, "reward_meter_std": 0.3483890891075134, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.500255286693573, "reward_total_composite_std": 0.3483890891075134, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 459.0} {"timestamp_utc": "2026-04-11T20:20:48Z", "mode": "train", "global_step": 460, "epoch": 0.017763361136855112, "loss": -0.0349, "grad_norm": 4.076127529144287, "learning_rate": 8.60909090909091e-06, "num_tokens": 987752.0, "completions/mean_length": 136.375, "completions/min_length": 119.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.375, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.5672988295555115, "rewards/meter/std": 0.4312629699707031, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5672988295555115, "rewards/total_composite/std": 0.4312629699707031, "reward": 0.5672988295555115, "reward_std": 0.4312629699707031, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.045063383877277374, "sampling/sampling_logp_difference/max": 2.0659866333007812, "sampling/importance_sampling_ratio/min": 0.1266932338476181, "sampling/importance_sampling_ratio/mean": 1.0044218301773071, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2906641475856304, "clip_ratio/low_mean": 0.017407751642167568, "clip_ratio/low_min": 0.017407751642167568, "clip_ratio/high_mean": 0.012001242313999683, "clip_ratio/high_max": 0.012001242313999683, "clip_ratio/region_mean": 0.02940899395616725, "reward_total_mean": 0.5672988295555115, "reward_meter_mean": 0.5672988295555115, "reward_meter_std": 0.4312629699707031, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5672988295555115, "reward_total_composite_std": 0.4312629699707031, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 460.0} {"timestamp_utc": "2026-04-11T20:20:55Z", "mode": "train", "global_step": 461, "epoch": 0.017801977139326536, "loss": 0.059, "grad_norm": 2.037872552871704, "learning_rate": 8.606060606060606e-06, "num_tokens": 991070.0, "completions/mean_length": 208.75, "completions/min_length": 192.0, "completions/max_length": 243.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 208.75, "completions/min_terminated_length": 192.0, "completions/max_terminated_length": 243.0, "rewards/meter/mean": 0.9929186105728149, "rewards/meter/std": 0.0022024763748049736, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9722763299942017, "rewards/total_composite/std": 0.0592639334499836, "reward": 0.9722763299942017, "reward_std": 0.059263937175273895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014968657866120338, "sampling/sampling_logp_difference/max": 1.7843563556671143, "sampling/importance_sampling_ratio/min": 0.16790510714054108, "sampling/importance_sampling_ratio/mean": 1.0022977590560913, "sampling/importance_sampling_ratio/max": 1.966139793395996, "entropy": 0.08123606955632567, "clip_ratio/low_mean": 0.002057613106444478, "clip_ratio/low_min": 0.002057613106444478, "clip_ratio/high_mean": 0.009281257749535143, "clip_ratio/high_max": 0.009281257749535143, "clip_ratio/region_mean": 0.011338870855979621, "reward_total_mean": 0.9722763299942017, "reward_meter_mean": 0.9929186105728149, "reward_meter_std": 0.0022024763748049736, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9722763299942017, "reward_total_composite_std": 0.0592639334499836, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 461.0} {"timestamp_utc": "2026-04-11T20:21:00Z", "mode": "train", "global_step": 462, "epoch": 0.01784059314179796, "loss": 0.0298, "grad_norm": 6.325527667999268, "learning_rate": 8.603030303030303e-06, "num_tokens": 993009.0, "completions/mean_length": 87.375, "completions/min_length": 71.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.375, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.82567298412323, "rewards/meter/std": 0.2936802804470062, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.82567298412323, "rewards/total_composite/std": 0.2936802804470062, "reward": 0.82567298412323, "reward_std": 0.29368025064468384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06957826763391495, "sampling/sampling_logp_difference/max": 1.0785102844238281, "sampling/importance_sampling_ratio/min": 0.34010180830955505, "sampling/importance_sampling_ratio/mean": 1.012847661972046, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.557247344404459, "clip_ratio/low_mean": 0.012249511666595936, "clip_ratio/low_min": 0.012249511666595936, "clip_ratio/high_mean": 0.05066513631027192, "clip_ratio/high_max": 0.05066513631027192, "clip_ratio/region_mean": 0.06291464797686785, "reward_total_mean": 0.82567298412323, "reward_meter_mean": 0.82567298412323, "reward_meter_std": 0.2936802804470062, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.82567298412323, "reward_total_composite_std": 0.2936802804470062, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 462.0} {"timestamp_utc": "2026-04-11T20:21:05Z", "mode": "train", "global_step": 463, "epoch": 0.017879209144269385, "loss": 0.0017, "grad_norm": 5.930828094482422, "learning_rate": 8.6e-06, "num_tokens": 994937.0, "completions/mean_length": 65.0, "completions/min_length": 64.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9922396540641785, "rewards/meter/std": 0.002804464427754283, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9922396540641785, "rewards/total_composite/std": 0.002804464427754283, "reward": 0.9922396540641785, "reward_std": 0.0028044627979397774, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033212851732969284, "sampling/sampling_logp_difference/max": 0.9475104808807373, "sampling/importance_sampling_ratio/min": 0.3877050578594208, "sampling/importance_sampling_ratio/mean": 1.003735065460205, "sampling/importance_sampling_ratio/max": 1.8714525699615479, "entropy": 0.1741385543718934, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.02107094577513635, "clip_ratio/high_max": 0.02107094577513635, "clip_ratio/region_mean": 0.024858824675902724, "reward_total_mean": 0.9922396540641785, "reward_meter_mean": 0.9922396540641785, "reward_meter_std": 0.002804464427754283, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9922396540641785, "reward_total_composite_std": 0.002804464427754283, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 463.0} {"timestamp_utc": "2026-04-11T20:21:15Z", "mode": "train", "global_step": 464, "epoch": 0.01791782514674081, "loss": 0.0244, "grad_norm": 2.3135251998901367, "learning_rate": 8.596969696969698e-06, "num_tokens": 996427.0, "completions/mean_length": 155.25, "completions/min_length": 22.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 36.333335876464844, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6161172389984131, "rewards/meter/std": 0.4752645790576935, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6161172389984131, "rewards/total_composite/std": 0.4752645790576935, "reward": 0.6161172389984131, "reward_std": 0.47526460886001587, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12042637169361115, "sampling/sampling_logp_difference/max": 1.1797246932983398, "sampling/importance_sampling_ratio/min": 0.307363361120224, "sampling/importance_sampling_ratio/mean": 0.9996461868286133, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.133492972701788, "clip_ratio/low_mean": 0.008184524020180106, "clip_ratio/low_min": 0.008184524020180106, "clip_ratio/high_mean": 0.08965541888028383, "clip_ratio/high_max": 0.08965541888028383, "clip_ratio/region_mean": 0.09783994290046394, "reward_total_mean": 0.6161172389984131, "reward_meter_mean": 0.6161172389984131, "reward_meter_std": 0.4752645790576935, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6161172389984131, "reward_total_composite_std": 0.4752645790576935, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 464.0} {"timestamp_utc": "2026-04-11T20:21:21Z", "mode": "train", "global_step": 465, "epoch": 0.017956441149212233, "loss": 0.0135, "grad_norm": 4.309906959533691, "learning_rate": 8.593939393939395e-06, "num_tokens": 998754.0, "completions/mean_length": 120.875, "completions/min_length": 113.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.875, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.8134843111038208, "rewards/meter/std": 0.3482135534286499, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6896439790725708, "rewards/total_composite/std": 0.4401980936527252, "reward": 0.6896439790725708, "reward_std": 0.44019806385040283, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03921005502343178, "sampling/sampling_logp_difference/max": 1.1664267778396606, "sampling/importance_sampling_ratio/min": 0.311477929353714, "sampling/importance_sampling_ratio/mean": 1.0006272792816162, "sampling/importance_sampling_ratio/max": 1.797261357307434, "entropy": 0.2514910716563463, "clip_ratio/low_mean": 0.015818335115909576, "clip_ratio/low_min": 0.015818335115909576, "clip_ratio/high_mean": 0.017931917682290077, "clip_ratio/high_max": 0.017931917682290077, "clip_ratio/region_mean": 0.033750252798199654, "reward_total_mean": 0.6896439790725708, "reward_meter_mean": 0.8134843111038208, "reward_meter_std": 0.3482135534286499, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6896439790725708, "reward_total_composite_std": 0.4401980936527252, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 465.0} {"timestamp_utc": "2026-04-11T20:21:27Z", "mode": "train", "global_step": 466, "epoch": 0.017995057151683657, "loss": 0.0569, "grad_norm": 5.061103820800781, "learning_rate": 8.590909090909092e-06, "num_tokens": 1000955.0, "completions/mean_length": 120.125, "completions/min_length": 112.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.125, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9295108914375305, "rewards/meter/std": 0.18052063882350922, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9295108914375305, "rewards/total_composite/std": 0.18052063882350922, "reward": 0.9295108914375305, "reward_std": 0.18052060902118683, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04497542977333069, "sampling/sampling_logp_difference/max": 2.3192155361175537, "sampling/importance_sampling_ratio/min": 0.09835071116685867, "sampling/importance_sampling_ratio/mean": 1.0039215087890625, "sampling/importance_sampling_ratio/max": 1.9984016418457031, "entropy": 0.291955066844821, "clip_ratio/low_mean": 0.006340579595416784, "clip_ratio/low_min": 0.006340579595416784, "clip_ratio/high_mean": 0.03716489998623729, "clip_ratio/high_max": 0.03716489998623729, "clip_ratio/region_mean": 0.04350547958165407, "reward_total_mean": 0.9295108914375305, "reward_meter_mean": 0.9295108914375305, "reward_meter_std": 0.18052063882350922, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9295108914375305, "reward_total_composite_std": 0.18052063882350922, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 466.0} {"timestamp_utc": "2026-04-11T20:21:33Z", "mode": "train", "global_step": 467, "epoch": 0.01803367315415508, "loss": 0.0437, "grad_norm": 2.9019546508789062, "learning_rate": 8.587878787878788e-06, "num_tokens": 1003675.0, "completions/mean_length": 152.0, "completions/min_length": 138.0, "completions/max_length": 174.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.0, "completions/min_terminated_length": 138.0, "completions/max_terminated_length": 174.0, "rewards/meter/mean": 0.8826419711112976, "rewards/meter/std": 0.06341774016618729, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8826419711112976, "rewards/total_composite/std": 0.06341774016618729, "reward": 0.8826419711112976, "reward_std": 0.06341774016618729, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015705382451415062, "sampling/sampling_logp_difference/max": 1.3150720596313477, "sampling/importance_sampling_ratio/min": 0.26845496892929077, "sampling/importance_sampling_ratio/mean": 1.0029544830322266, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0712776998989284, "clip_ratio/low_mean": 0.00698582676704973, "clip_ratio/low_min": 0.00698582676704973, "clip_ratio/high_mean": 0.008227241458371282, "clip_ratio/high_max": 0.008227241458371282, "clip_ratio/region_mean": 0.015213068225421011, "reward_total_mean": 0.8826419711112976, "reward_meter_mean": 0.8826419711112976, "reward_meter_std": 0.06341774016618729, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8826419711112976, "reward_total_composite_std": 0.06341774016618729, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 467.0} {"timestamp_utc": "2026-04-11T20:21:42Z", "mode": "train", "global_step": 468, "epoch": 0.018072289156626505, "loss": -0.0241, "grad_norm": 0.9965296983718872, "learning_rate": 8.584848484848485e-06, "num_tokens": 1008541.0, "completions/mean_length": 386.25, "completions/min_length": 359.0, "completions/max_length": 413.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 386.25, "completions/min_terminated_length": 359.0, "completions/max_terminated_length": 413.0, "rewards/meter/mean": 0.9619865417480469, "rewards/meter/std": 0.025031376630067825, "rewards/count_adherence/mean": 0.9431818723678589, "rewards/count_adherence/std": 0.047049909830093384, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.906877875328064, "rewards/total_composite/std": 0.04043539986014366, "reward": 0.906877875328064, "reward_std": 0.04043539986014366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00543476827442646, "sampling/sampling_logp_difference/max": 1.920891523361206, "sampling/importance_sampling_ratio/min": 0.4794699549674988, "sampling/importance_sampling_ratio/mean": 1.0018101930618286, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.028902734396979213, "clip_ratio/low_mean": 0.0020095510117243975, "clip_ratio/low_min": 0.0020095510117243975, "clip_ratio/high_mean": 0.0006329113966785371, "clip_ratio/high_max": 0.0006329113966785371, "clip_ratio/region_mean": 0.0026424624084029347, "reward_total_mean": 0.906877875328064, "reward_meter_mean": 0.9619865417480469, "reward_meter_std": 0.025031376630067825, "reward_count_adherence_mean": 0.9431818723678589, "reward_count_adherence_std": 0.047049909830093384, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.906877875328064, "reward_total_composite_std": 0.04043539986014366, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 468.0} {"timestamp_utc": "2026-04-11T20:21:47Z", "mode": "train", "global_step": 469, "epoch": 0.01811090515909793, "loss": 0.1817, "grad_norm": 7.141157150268555, "learning_rate": 8.581818181818183e-06, "num_tokens": 1010085.0, "completions/mean_length": 37.0, "completions/min_length": 30.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.857481062412262, "rewards/meter/std": 0.2525491416454315, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7328898906707764, "rewards/total_composite/std": 0.38510990142822266, "reward": 0.7328898906707764, "reward_std": 0.38510987162590027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04425249993801117, "sampling/sampling_logp_difference/max": 1.5494416952133179, "sampling/importance_sampling_ratio/min": 0.2123665064573288, "sampling/importance_sampling_ratio/mean": 0.9989219903945923, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17447292152792215, "clip_ratio/low_mean": 0.009848485235124826, "clip_ratio/low_min": 0.009848485235124826, "clip_ratio/high_mean": 0.026214548386633396, "clip_ratio/high_max": 0.026214548386633396, "clip_ratio/region_mean": 0.03606303362175822, "reward_total_mean": 0.7328898906707764, "reward_meter_mean": 0.857481062412262, "reward_meter_std": 0.2525491416454315, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7328898906707764, "reward_total_composite_std": 0.38510990142822266, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 469.0} {"timestamp_utc": "2026-04-11T20:21:52Z", "mode": "train", "global_step": 470, "epoch": 0.018149521161569353, "loss": 0.0294, "grad_norm": 7.436746597290039, "learning_rate": 8.57878787878788e-06, "num_tokens": 1012136.0, "completions/mean_length": 96.375, "completions/min_length": 94.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.375, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.9915547370910645, "rewards/meter/std": 0.0038915486074984074, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9915547370910645, "rewards/total_composite/std": 0.0038915486074984074, "reward": 0.9915547370910645, "reward_std": 0.0038915553595870733, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038666851818561554, "sampling/sampling_logp_difference/max": 1.3998184204101562, "sampling/importance_sampling_ratio/min": 0.24664175510406494, "sampling/importance_sampling_ratio/mean": 1.0057390928268433, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22278593480587006, "clip_ratio/low_mean": 0.007512626354582608, "clip_ratio/low_min": 0.007512626354582608, "clip_ratio/high_mean": 0.0195100650889799, "clip_ratio/high_max": 0.0195100650889799, "clip_ratio/region_mean": 0.027022691443562508, "reward_total_mean": 0.9915547370910645, "reward_meter_mean": 0.9915547370910645, "reward_meter_std": 0.0038915486074984074, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9915547370910645, "reward_total_composite_std": 0.0038915486074984074, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 470.0} {"timestamp_utc": "2026-04-11T20:21:57Z", "mode": "train", "global_step": 471, "epoch": 0.018188137164040778, "loss": 0.0586, "grad_norm": 15.974438667297363, "learning_rate": 8.575757575757575e-06, "num_tokens": 1013700.0, "completions/mean_length": 40.5, "completions/min_length": 28.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.38729679584503174, "rewards/meter/std": 0.3866852819919586, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.38729679584503174, "rewards/total_composite/std": 0.3866852819919586, "reward": 0.38729679584503174, "reward_std": 0.3866852819919586, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16885001957416534, "sampling/sampling_logp_difference/max": 1.9867620468139648, "sampling/importance_sampling_ratio/min": 0.1371387541294098, "sampling/importance_sampling_ratio/mean": 1.0078424215316772, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0023160837590694, "clip_ratio/low_mean": 0.07797870878130198, "clip_ratio/low_min": 0.07797870878130198, "clip_ratio/high_mean": 0.09589226730167866, "clip_ratio/high_max": 0.09589226730167866, "clip_ratio/region_mean": 0.17387097608298063, "reward_total_mean": 0.38729679584503174, "reward_meter_mean": 0.38729679584503174, "reward_meter_std": 0.3866852819919586, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.38729679584503174, "reward_total_composite_std": 0.3866852819919586, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 471.0} {"timestamp_utc": "2026-04-11T20:22:05Z", "mode": "train", "global_step": 472, "epoch": 0.0182267531665122, "loss": -0.0246, "grad_norm": 1.0963542461395264, "learning_rate": 8.572727272727274e-06, "num_tokens": 1017570.0, "completions/mean_length": 278.75, "completions/min_length": 241.0, "completions/max_length": 334.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 278.75, "completions/min_terminated_length": 241.0, "completions/max_terminated_length": 334.0, "rewards/meter/mean": 0.924202561378479, "rewards/meter/std": 0.0937681496143341, "rewards/count_adherence/mean": 0.910714328289032, "rewards/count_adherence/std": 0.07393559068441391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8407608866691589, "rewards/total_composite/std": 0.10190222412347794, "reward": 0.8407608866691589, "reward_std": 0.10190220922231674, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008891636505723, "sampling/sampling_logp_difference/max": 0.992754340171814, "sampling/importance_sampling_ratio/min": 0.370554655790329, "sampling/importance_sampling_ratio/mean": 1.000677227973938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03723532357253134, "clip_ratio/low_mean": 0.004235844942741096, "clip_ratio/low_min": 0.004235844942741096, "clip_ratio/high_mean": 0.004306173766963184, "clip_ratio/high_max": 0.004306173766963184, "clip_ratio/region_mean": 0.00854201870970428, "reward_total_mean": 0.8407608866691589, "reward_meter_mean": 0.924202561378479, "reward_meter_std": 0.0937681496143341, "reward_count_adherence_mean": 0.910714328289032, "reward_count_adherence_std": 0.07393559068441391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8407608866691589, "reward_total_composite_std": 0.10190222412347794, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 472.0} {"timestamp_utc": "2026-04-11T20:22:11Z", "mode": "train", "global_step": 473, "epoch": 0.018265369168983626, "loss": -0.0118, "grad_norm": 5.147512912750244, "learning_rate": 8.56969696969697e-06, "num_tokens": 1020348.0, "completions/mean_length": 171.25, "completions/min_length": 149.0, "completions/max_length": 196.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 171.25, "completions/min_terminated_length": 149.0, "completions/max_terminated_length": 196.0, "rewards/meter/mean": 0.5626472234725952, "rewards/meter/std": 0.38277679681777954, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.49889272451400757, "rewards/total_composite/std": 0.33463042974472046, "reward": 0.49889272451400757, "reward_std": 0.33463039994239807, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.053236186504364014, "sampling/sampling_logp_difference/max": 2.106261730194092, "sampling/importance_sampling_ratio/min": 0.12169203162193298, "sampling/importance_sampling_ratio/mean": 0.9987762570381165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17230255994945765, "clip_ratio/low_mean": 0.01952884637285024, "clip_ratio/low_min": 0.01952884637285024, "clip_ratio/high_mean": 0.01820858521386981, "clip_ratio/high_max": 0.01820858521386981, "clip_ratio/region_mean": 0.03773743158672005, "reward_total_mean": 0.49889272451400757, "reward_meter_mean": 0.5626472234725952, "reward_meter_std": 0.38277679681777954, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.49889272451400757, "reward_total_composite_std": 0.33463042974472046, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 473.0} {"timestamp_utc": "2026-04-11T20:22:19Z", "mode": "train", "global_step": 474, "epoch": 0.01830398517145505, "loss": 0.0284, "grad_norm": 2.0184075832366943, "learning_rate": 8.566666666666667e-06, "num_tokens": 1024089.0, "completions/mean_length": 253.625, "completions/min_length": 223.0, "completions/max_length": 291.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 253.625, "completions/min_terminated_length": 223.0, "completions/max_terminated_length": 291.0, "rewards/meter/mean": 0.998430073261261, "rewards/meter/std": 0.0012139956234022975, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.915224552154541, "rewards/total_composite/std": 0.08891873806715012, "reward": 0.915224552154541, "reward_std": 0.08891873806715012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020224176347255707, "sampling/sampling_logp_difference/max": 2.128042697906494, "sampling/importance_sampling_ratio/min": 0.1190701276063919, "sampling/importance_sampling_ratio/mean": 0.9997615814208984, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10501971561461687, "clip_ratio/low_mean": 0.005101353977806866, "clip_ratio/low_min": 0.005101353977806866, "clip_ratio/high_mean": 0.004681251273723319, "clip_ratio/high_max": 0.004681251273723319, "clip_ratio/region_mean": 0.009782605251530185, "reward_total_mean": 0.915224552154541, "reward_meter_mean": 0.998430073261261, "reward_meter_std": 0.0012139956234022975, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.915224552154541, "reward_total_composite_std": 0.08891873806715012, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 474.0} {"timestamp_utc": "2026-04-11T20:22:24Z", "mode": "train", "global_step": 475, "epoch": 0.018342601173926474, "loss": 0.0063, "grad_norm": 5.0996994972229, "learning_rate": 8.563636363636364e-06, "num_tokens": 1025987.0, "completions/mean_length": 71.25, "completions/min_length": 65.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.25, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8450901508331299, "rewards/meter/std": 0.1726076304912567, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8450901508331299, "rewards/total_composite/std": 0.1726076304912567, "reward": 0.8450901508331299, "reward_std": 0.17260761559009552, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.043718110769987106, "sampling/sampling_logp_difference/max": 1.1217999458312988, "sampling/importance_sampling_ratio/min": 0.3256930410861969, "sampling/importance_sampling_ratio/mean": 0.9991965889930725, "sampling/importance_sampling_ratio/max": 1.6401515007019043, "entropy": 0.2513603400439024, "clip_ratio/low_mean": 0.006897549610584974, "clip_ratio/low_min": 0.006897549610584974, "clip_ratio/high_mean": 0.04206577688455582, "clip_ratio/high_max": 0.04206577688455582, "clip_ratio/region_mean": 0.04896332649514079, "reward_total_mean": 0.8450901508331299, "reward_meter_mean": 0.8450901508331299, "reward_meter_std": 0.1726076304912567, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8450901508331299, "reward_total_composite_std": 0.1726076304912567, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 475.0} {"timestamp_utc": "2026-04-11T20:22:28Z", "mode": "train", "global_step": 476, "epoch": 0.018381217176397898, "loss": -0.0343, "grad_norm": 7.3300299644470215, "learning_rate": 8.560606060606062e-06, "num_tokens": 1027490.0, "completions/mean_length": 29.875, "completions/min_length": 27.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.875, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9858591556549072, "rewards/meter/std": 0.004257791675627232, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9858591556549072, "rewards/total_composite/std": 0.004257791675627232, "reward": 0.9858591556549072, "reward_std": 0.004257792141288519, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03374743461608887, "sampling/sampling_logp_difference/max": 0.8543341159820557, "sampling/importance_sampling_ratio/min": 0.5488837957382202, "sampling/importance_sampling_ratio/mean": 1.0109765529632568, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20345518365502357, "clip_ratio/low_mean": 0.009093915577977896, "clip_ratio/low_min": 0.009093915577977896, "clip_ratio/high_mean": 0.012941297609359026, "clip_ratio/high_max": 0.012941297609359026, "clip_ratio/region_mean": 0.02203521318733692, "reward_total_mean": 0.9858591556549072, "reward_meter_mean": 0.9858591556549072, "reward_meter_std": 0.004257791675627232, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9858591556549072, "reward_total_composite_std": 0.004257791675627232, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 476.0} {"timestamp_utc": "2026-04-11T20:22:34Z", "mode": "train", "global_step": 477, "epoch": 0.018419833178869322, "loss": 0.018, "grad_norm": 3.7667415142059326, "learning_rate": 8.557575757575757e-06, "num_tokens": 1029622.0, "completions/mean_length": 108.5, "completions/min_length": 102.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.5, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9522513151168823, "rewards/meter/std": 0.056059498339891434, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9522513151168823, "rewards/total_composite/std": 0.056059498339891434, "reward": 0.9522513151168823, "reward_std": 0.05605950206518173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026357771828770638, "sampling/sampling_logp_difference/max": 1.1791865825653076, "sampling/importance_sampling_ratio/min": 0.3075287938117981, "sampling/importance_sampling_ratio/mean": 1.0057148933410645, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15065084025263786, "clip_ratio/low_mean": 0.005435622064396739, "clip_ratio/low_min": 0.005435622064396739, "clip_ratio/high_mean": 0.011585089145228267, "clip_ratio/high_max": 0.011585089145228267, "clip_ratio/region_mean": 0.017020711209625006, "reward_total_mean": 0.9522513151168823, "reward_meter_mean": 0.9522513151168823, "reward_meter_std": 0.056059498339891434, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9522513151168823, "reward_total_composite_std": 0.056059498339891434, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 477.0} {"timestamp_utc": "2026-04-11T20:22:38Z", "mode": "train", "global_step": 478, "epoch": 0.018458449181340746, "loss": -0.0113, "grad_norm": 10.037378311157227, "learning_rate": 8.554545454545456e-06, "num_tokens": 1031189.0, "completions/mean_length": 37.875, "completions/min_length": 35.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.978717565536499, "rewards/meter/std": 0.019986465573310852, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.978717565536499, "rewards/total_composite/std": 0.019986465573310852, "reward": 0.978717565536499, "reward_std": 0.019986478611826897, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08604246377944946, "sampling/sampling_logp_difference/max": 1.4360220432281494, "sampling/importance_sampling_ratio/min": 0.23787212371826172, "sampling/importance_sampling_ratio/mean": 1.0022932291030884, "sampling/importance_sampling_ratio/max": 1.5463435649871826, "entropy": 0.5912935622036457, "clip_ratio/low_mean": 0.02712087077088654, "clip_ratio/low_min": 0.02712087077088654, "clip_ratio/high_mean": 0.04250439163297415, "clip_ratio/high_max": 0.04250439163297415, "clip_ratio/region_mean": 0.06962526240386069, "reward_total_mean": 0.978717565536499, "reward_meter_mean": 0.978717565536499, "reward_meter_std": 0.019986465573310852, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.978717565536499, "reward_total_composite_std": 0.019986465573310852, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 478.0} {"timestamp_utc": "2026-04-11T20:22:43Z", "mode": "train", "global_step": 479, "epoch": 0.01849706518381217, "loss": 0.0424, "grad_norm": 10.624458312988281, "learning_rate": 8.551515151515152e-06, "num_tokens": 1032963.0, "completions/mean_length": 39.75, "completions/min_length": 37.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.75, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9893389940261841, "rewards/meter/std": 0.015485532581806183, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9893389940261841, "rewards/total_composite/std": 0.015485532581806183, "reward": 0.9893389940261841, "reward_std": 0.015485531650483608, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10419061034917831, "sampling/sampling_logp_difference/max": 1.3149088621139526, "sampling/importance_sampling_ratio/min": 0.2684987783432007, "sampling/importance_sampling_ratio/mean": 1.0154733657836914, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5402822978794575, "clip_ratio/low_mean": 0.018391148187220097, "clip_ratio/low_min": 0.018391148187220097, "clip_ratio/high_mean": 0.05710198846645653, "clip_ratio/high_max": 0.05710198846645653, "clip_ratio/region_mean": 0.07549313665367663, "reward_total_mean": 0.9893389940261841, "reward_meter_mean": 0.9893389940261841, "reward_meter_std": 0.015485532581806183, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9893389940261841, "reward_total_composite_std": 0.015485532581806183, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 479.0} {"timestamp_utc": "2026-04-11T20:22:48Z", "mode": "train", "global_step": 480, "epoch": 0.018535681186283594, "loss": -0.0113, "grad_norm": 20.665754318237305, "learning_rate": 8.548484848484849e-06, "num_tokens": 1034496.0, "completions/mean_length": 31.625, "completions/min_length": 25.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9100282192230225, "rewards/meter/std": 0.14061063528060913, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9100282192230225, "rewards/total_composite/std": 0.14061063528060913, "reward": 0.9100282192230225, "reward_std": 0.14061065018177032, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13002178072929382, "sampling/sampling_logp_difference/max": 0.977691650390625, "sampling/importance_sampling_ratio/min": 0.37617847323417664, "sampling/importance_sampling_ratio/mean": 1.0112000703811646, "sampling/importance_sampling_ratio/max": 1.649492859840393, "entropy": 1.2414276078343391, "clip_ratio/low_mean": 0.05937229562550783, "clip_ratio/low_min": 0.05937229562550783, "clip_ratio/high_mean": 0.05253493972122669, "clip_ratio/high_max": 0.05253493972122669, "clip_ratio/region_mean": 0.11190723534673452, "reward_total_mean": 0.9100282192230225, "reward_meter_mean": 0.9100282192230225, "reward_meter_std": 0.14061063528060913, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9100282192230225, "reward_total_composite_std": 0.14061063528060913, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 480.0} {"timestamp_utc": "2026-04-11T20:22:53Z", "mode": "train", "global_step": 481, "epoch": 0.01857429718875502, "loss": 0.024, "grad_norm": 8.993289947509766, "learning_rate": 8.545454545454546e-06, "num_tokens": 1036446.0, "completions/mean_length": 76.75, "completions/min_length": 58.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9631932973861694, "rewards/meter/std": 0.08677735924720764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9631932973861694, "rewards/total_composite/std": 0.08677735924720764, "reward": 0.9631932973861694, "reward_std": 0.08677736669778824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0779096931219101, "sampling/sampling_logp_difference/max": 1.344578742980957, "sampling/importance_sampling_ratio/min": 0.2606494724750519, "sampling/importance_sampling_ratio/mean": 0.9976180195808411, "sampling/importance_sampling_ratio/max": 1.636481761932373, "entropy": 0.5781661160290241, "clip_ratio/low_mean": 0.010937499813735485, "clip_ratio/low_min": 0.010937499813735485, "clip_ratio/high_mean": 0.08672621939331293, "clip_ratio/high_max": 0.08672621939331293, "clip_ratio/region_mean": 0.09766371920704842, "reward_total_mean": 0.9631932973861694, "reward_meter_mean": 0.9631932973861694, "reward_meter_std": 0.08677735924720764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9631932973861694, "reward_total_composite_std": 0.08677735924720764, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 481.0} {"timestamp_utc": "2026-04-11T20:23:03Z", "mode": "train", "global_step": 482, "epoch": 0.018612913191226443, "loss": -0.244, "grad_norm": 1.2208207845687866, "learning_rate": 8.542424242424243e-06, "num_tokens": 1040023.0, "completions/mean_length": 300.125, "completions/min_length": 254.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 269.8571472167969, "completions/min_terminated_length": 254.0, "completions/max_terminated_length": 288.0, "rewards/meter/mean": 0.9346262216567993, "rewards/meter/std": 0.16473090648651123, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.29574236273765564, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7929118871688843, "rewards/total_composite/std": 0.3079202473163605, "reward": 0.7929118871688843, "reward_std": 0.3079202473163605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012091459706425667, "sampling/sampling_logp_difference/max": 0.9693708419799805, "sampling/importance_sampling_ratio/min": 0.3793216049671173, "sampling/importance_sampling_ratio/mean": 1.0012937784194946, "sampling/importance_sampling_ratio/max": 1.6571389436721802, "entropy": 0.07355018192902207, "clip_ratio/low_mean": 0.002808988792821765, "clip_ratio/low_min": 0.002808988792821765, "clip_ratio/high_mean": 0.005991354206344113, "clip_ratio/high_max": 0.005991354206344113, "clip_ratio/region_mean": 0.008800342999165878, "reward_total_mean": 0.7929118871688843, "reward_meter_mean": 0.9346262216567993, "reward_meter_std": 0.16473090648651123, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.29574236273765564, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7929118871688843, "reward_total_composite_std": 0.3079202473163605, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 482.0} {"timestamp_utc": "2026-04-11T20:23:09Z", "mode": "train", "global_step": 483, "epoch": 0.018651529193697867, "loss": 0.0021, "grad_norm": 3.103454828262329, "learning_rate": 8.539393939393939e-06, "num_tokens": 1043145.0, "completions/mean_length": 180.25, "completions/min_length": 161.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 180.25, "completions/min_terminated_length": 161.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9860587120056152, "rewards/meter/std": 0.00629128934815526, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9860587120056152, "rewards/total_composite/std": 0.00629128934815526, "reward": 0.9860587120056152, "reward_std": 0.006291288882493973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011846227571368217, "sampling/sampling_logp_difference/max": 0.8737516403198242, "sampling/importance_sampling_ratio/min": 0.41738274693489075, "sampling/importance_sampling_ratio/mean": 1.0001847743988037, "sampling/importance_sampling_ratio/max": 1.7320133447647095, "entropy": 0.0649343291297555, "clip_ratio/low_mean": 0.0014594576205126941, "clip_ratio/low_min": 0.0014594576205126941, "clip_ratio/high_mean": 0.010245901241432875, "clip_ratio/high_max": 0.010245901241432875, "clip_ratio/region_mean": 0.01170535886194557, "reward_total_mean": 0.9860587120056152, "reward_meter_mean": 0.9860587120056152, "reward_meter_std": 0.00629128934815526, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9860587120056152, "reward_total_composite_std": 0.00629128934815526, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 483.0} {"timestamp_utc": "2026-04-11T20:23:14Z", "mode": "train", "global_step": 484, "epoch": 0.01869014519616929, "loss": 0.0128, "grad_norm": 9.362083435058594, "learning_rate": 8.536363636363636e-06, "num_tokens": 1044895.0, "completions/mean_length": 66.75, "completions/min_length": 50.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.75, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.7406154870986938, "rewards/meter/std": 0.36668860912323, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6785883903503418, "rewards/total_composite/std": 0.3599132299423218, "reward": 0.6785883903503418, "reward_std": 0.35991325974464417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10126848518848419, "sampling/sampling_logp_difference/max": 1.3694827556610107, "sampling/importance_sampling_ratio/min": 0.25423842668533325, "sampling/importance_sampling_ratio/mean": 1.0062280893325806, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9045082405209541, "clip_ratio/low_mean": 0.029679158004000783, "clip_ratio/low_min": 0.029679158004000783, "clip_ratio/high_mean": 0.059974749106913805, "clip_ratio/high_max": 0.059974749106913805, "clip_ratio/region_mean": 0.08965390711091459, "reward_total_mean": 0.6785883903503418, "reward_meter_mean": 0.7406154870986938, "reward_meter_std": 0.36668860912323, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6785883903503418, "reward_total_composite_std": 0.3599132299423218, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 484.0} {"timestamp_utc": "2026-04-11T20:23:19Z", "mode": "train", "global_step": 485, "epoch": 0.018728761198640715, "loss": -0.0278, "grad_norm": 11.369089126586914, "learning_rate": 8.533333333333335e-06, "num_tokens": 1046762.0, "completions/mean_length": 61.375, "completions/min_length": 49.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8861362338066101, "rewards/meter/std": 0.2756871283054352, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8861362338066101, "rewards/total_composite/std": 0.2756871283054352, "reward": 0.8861362338066101, "reward_std": 0.2756870985031128, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08699135482311249, "sampling/sampling_logp_difference/max": 1.3300657272338867, "sampling/importance_sampling_ratio/min": 0.26445987820625305, "sampling/importance_sampling_ratio/mean": 0.9978853464126587, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5880453810095787, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/high_mean": 0.07197061297483742, "clip_ratio/high_max": 0.07197061297483742, "clip_ratio/region_mean": 0.08759561297483742, "reward_total_mean": 0.8861362338066101, "reward_meter_mean": 0.8861362338066101, "reward_meter_std": 0.2756871283054352, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8861362338066101, "reward_total_composite_std": 0.2756871283054352, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 485.0} {"timestamp_utc": "2026-04-11T20:23:24Z", "mode": "train", "global_step": 486, "epoch": 0.018767377201112143, "loss": -0.026, "grad_norm": 4.333034992218018, "learning_rate": 8.53030303030303e-06, "num_tokens": 1048822.0, "completions/mean_length": 99.5, "completions/min_length": 84.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.5, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.786818265914917, "rewards/meter/std": 0.3528897166252136, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7474471926689148, "rewards/total_composite/std": 0.35029318928718567, "reward": 0.7474471926689148, "reward_std": 0.3502931594848633, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0431230328977108, "sampling/sampling_logp_difference/max": 1.4832372665405273, "sampling/importance_sampling_ratio/min": 0.22690196335315704, "sampling/importance_sampling_ratio/mean": 1.002170443534851, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2146557718515396, "clip_ratio/low_mean": 0.02085444121621549, "clip_ratio/low_min": 0.02085444121621549, "clip_ratio/high_mean": 0.030109542072750628, "clip_ratio/high_max": 0.030109542072750628, "clip_ratio/region_mean": 0.05096398328896612, "reward_total_mean": 0.7474471926689148, "reward_meter_mean": 0.786818265914917, "reward_meter_std": 0.3528897166252136, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7474471926689148, "reward_total_composite_std": 0.35029318928718567, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 486.0} {"timestamp_utc": "2026-04-11T20:23:33Z", "mode": "train", "global_step": 487, "epoch": 0.018805993203583567, "loss": -0.0076, "grad_norm": 1.3781453371047974, "learning_rate": 8.527272727272728e-06, "num_tokens": 1053671.0, "completions/mean_length": 400.125, "completions/min_length": 344.0, "completions/max_length": 439.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 400.125, "completions/min_terminated_length": 344.0, "completions/max_terminated_length": 439.0, "rewards/meter/mean": 0.9634721279144287, "rewards/meter/std": 0.0763736218214035, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6021700501441956, "rewards/total_composite/std": 0.047733522951602936, "reward": 0.6021700501441956, "reward_std": 0.047733522951602936, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018493853509426117, "sampling/sampling_logp_difference/max": 1.362187385559082, "sampling/importance_sampling_ratio/min": 0.25609996914863586, "sampling/importance_sampling_ratio/mean": 1.0029096603393555, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12399658421054482, "clip_ratio/low_mean": 0.002577319508418441, "clip_ratio/low_min": 0.002577319508418441, "clip_ratio/high_mean": 0.013687330298125744, "clip_ratio/high_max": 0.013687330298125744, "clip_ratio/region_mean": 0.016264649806544185, "reward_total_mean": 0.6021700501441956, "reward_meter_mean": 0.9634721279144287, "reward_meter_std": 0.0763736218214035, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6021700501441956, "reward_total_composite_std": 0.047733522951602936, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 487.0} {"timestamp_utc": "2026-04-11T20:23:38Z", "mode": "train", "global_step": 488, "epoch": 0.01884460920605499, "loss": -0.0275, "grad_norm": 12.93553352355957, "learning_rate": 8.524242424242425e-06, "num_tokens": 1055530.0, "completions/mean_length": 56.375, "completions/min_length": 50.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9578395485877991, "rewards/meter/std": 0.06491218507289886, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9578395485877991, "rewards/total_composite/std": 0.06491218507289886, "reward": 0.9578395485877991, "reward_std": 0.06491218507289886, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11469703167676926, "sampling/sampling_logp_difference/max": 1.230273962020874, "sampling/importance_sampling_ratio/min": 0.29221248626708984, "sampling/importance_sampling_ratio/mean": 1.005127191543579, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9573509357869625, "clip_ratio/low_mean": 0.019999999552965164, "clip_ratio/low_min": 0.019999999552965164, "clip_ratio/high_mean": 0.08754385984502733, "clip_ratio/high_max": 0.08754385984502733, "clip_ratio/region_mean": 0.10754385939799249, "reward_total_mean": 0.9578395485877991, "reward_meter_mean": 0.9578395485877991, "reward_meter_std": 0.06491218507289886, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9578395485877991, "reward_total_composite_std": 0.06491218507289886, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 488.0} {"timestamp_utc": "2026-04-11T20:23:46Z", "mode": "train", "global_step": 489, "epoch": 0.018883225208526415, "loss": 0.0046, "grad_norm": 3.998319625854492, "learning_rate": 8.521212121212123e-06, "num_tokens": 1058017.0, "completions/mean_length": 133.875, "completions/min_length": 120.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.875, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.9484606981277466, "rewards/meter/std": 0.09933194518089294, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9484606981277466, "rewards/total_composite/std": 0.09933194518089294, "reward": 0.9484606981277466, "reward_std": 0.09933193773031235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0375179685652256, "sampling/sampling_logp_difference/max": 1.4298303127288818, "sampling/importance_sampling_ratio/min": 0.23934954404830933, "sampling/importance_sampling_ratio/mean": 1.006198763847351, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25978855788707733, "clip_ratio/low_mean": 0.00903480825945735, "clip_ratio/low_min": 0.00903480825945735, "clip_ratio/high_mean": 0.02726867760065943, "clip_ratio/high_max": 0.02726867760065943, "clip_ratio/region_mean": 0.03630348586011678, "reward_total_mean": 0.9484606981277466, "reward_meter_mean": 0.9484606981277466, "reward_meter_std": 0.09933194518089294, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9484606981277466, "reward_total_composite_std": 0.09933194518089294, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 489.0} {"timestamp_utc": "2026-04-11T20:23:56Z", "mode": "train", "global_step": 490, "epoch": 0.01892184121099784, "loss": 0.1271, "grad_norm": 0.38429051637649536, "learning_rate": 8.518181818181818e-06, "num_tokens": 1060315.0, "completions/mean_length": 510.25, "completions/min_length": 498.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.875, "completions/mean_terminated_length": 498.0, "completions/min_terminated_length": 498.0, "completions/max_terminated_length": 498.0, "rewards/meter/mean": 0.9891150593757629, "rewards/meter/std": 0.00664390018209815, "rewards/count_adherence/mean": 0.8557692170143127, "rewards/count_adherence/std": 0.08661473542451859, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8464405536651611, "rewards/total_composite/std": 0.08564455062150955, "reward": 0.8464405536651611, "reward_std": 0.08564455062150955, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011192544363439083, "sampling/sampling_logp_difference/max": 0.9943814277648926, "sampling/importance_sampling_ratio/min": 0.3699522316455841, "sampling/importance_sampling_ratio/mean": 0.9997017979621887, "sampling/importance_sampling_ratio/max": 1.7250312566757202, "entropy": 0.00822315365076065, "clip_ratio/low_mean": 0.0007530120201408863, "clip_ratio/low_min": 0.0007530120201408863, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0007530120201408863, "reward_total_mean": 0.8464405536651611, "reward_meter_mean": 0.9891150593757629, "reward_meter_std": 0.00664390018209815, "reward_count_adherence_mean": 0.8557692170143127, "reward_count_adherence_std": 0.08661473542451859, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8464405536651611, "reward_total_composite_std": 0.08564455062150955, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 490.0} {"timestamp_utc": "2026-04-11T20:24:03Z", "mode": "train", "global_step": 491, "epoch": 0.018960457213469263, "loss": 0.0351, "grad_norm": 4.761613368988037, "learning_rate": 8.515151515151517e-06, "num_tokens": 1062421.0, "completions/mean_length": 93.25, "completions/min_length": 61.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.6782705783843994, "rewards/meter/std": 0.363802045583725, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6782705783843994, "rewards/total_composite/std": 0.363802045583725, "reward": 0.6782705783843994, "reward_std": 0.363802045583725, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08261455595493317, "sampling/sampling_logp_difference/max": 1.3461203575134277, "sampling/importance_sampling_ratio/min": 0.26024797558784485, "sampling/importance_sampling_ratio/mean": 1.0159639120101929, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.900152787566185, "clip_ratio/low_mean": 0.043553632916882634, "clip_ratio/low_min": 0.043553632916882634, "clip_ratio/high_mean": 0.03605138626880944, "clip_ratio/high_max": 0.03605138626880944, "clip_ratio/region_mean": 0.07960501918569207, "reward_total_mean": 0.6782705783843994, "reward_meter_mean": 0.6782705783843994, "reward_meter_std": 0.363802045583725, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6782705783843994, "reward_total_composite_std": 0.363802045583725, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 491.0} {"timestamp_utc": "2026-04-11T20:24:08Z", "mode": "train", "global_step": 492, "epoch": 0.018999073215940687, "loss": 0.0363, "grad_norm": 10.827360153198242, "learning_rate": 8.512121212121213e-06, "num_tokens": 1064305.0, "completions/mean_length": 68.5, "completions/min_length": 59.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.5, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9250147342681885, "rewards/meter/std": 0.151312917470932, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9250147342681885, "rewards/total_composite/std": 0.151312917470932, "reward": 0.9250147342681885, "reward_std": 0.1513129323720932, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10603255778551102, "sampling/sampling_logp_difference/max": 1.867074966430664, "sampling/importance_sampling_ratio/min": 0.1545751392841339, "sampling/importance_sampling_ratio/mean": 1.018401861190796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7679142504930496, "clip_ratio/low_mean": 0.032360111363232136, "clip_ratio/low_min": 0.032360111363232136, "clip_ratio/high_mean": 0.03652191942092031, "clip_ratio/high_max": 0.03652191942092031, "clip_ratio/region_mean": 0.06888203078415245, "reward_total_mean": 0.9250147342681885, "reward_meter_mean": 0.9250147342681885, "reward_meter_std": 0.151312917470932, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9250147342681885, "reward_total_composite_std": 0.151312917470932, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 492.0} {"timestamp_utc": "2026-04-11T20:24:13Z", "mode": "train", "global_step": 493, "epoch": 0.01903768921841211, "loss": -0.0015, "grad_norm": 5.6415114402771, "learning_rate": 8.50909090909091e-06, "num_tokens": 1066128.0, "completions/mean_length": 68.875, "completions/min_length": 53.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9896060228347778, "rewards/meter/std": 0.009286360815167427, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9896060228347778, "rewards/total_composite/std": 0.009286360815167427, "reward": 0.9896060228347778, "reward_std": 0.009286360815167427, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07539483904838562, "sampling/sampling_logp_difference/max": 1.9518604278564453, "sampling/importance_sampling_ratio/min": 0.14200963079929352, "sampling/importance_sampling_ratio/mean": 1.0087600946426392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5212747771292925, "clip_ratio/low_mean": 0.016476215794682503, "clip_ratio/low_min": 0.016476215794682503, "clip_ratio/high_mean": 0.039439259795472026, "clip_ratio/high_max": 0.039439259795472026, "clip_ratio/region_mean": 0.05591547559015453, "reward_total_mean": 0.9896060228347778, "reward_meter_mean": 0.9896060228347778, "reward_meter_std": 0.009286360815167427, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9896060228347778, "reward_total_composite_std": 0.009286360815167427, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 493.0} {"timestamp_utc": "2026-04-11T20:24:17Z", "mode": "train", "global_step": 494, "epoch": 0.019076305220883535, "loss": 0.0225, "grad_norm": 6.613814353942871, "learning_rate": 8.506060606060607e-06, "num_tokens": 1067822.0, "completions/mean_length": 58.75, "completions/min_length": 54.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9344394207000732, "rewards/meter/std": 0.060838740319013596, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9344394207000732, "rewards/total_composite/std": 0.060838740319013596, "reward": 0.9344394207000732, "reward_std": 0.060838740319013596, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06783849745988846, "sampling/sampling_logp_difference/max": 1.9469289779663086, "sampling/importance_sampling_ratio/min": 0.14271166920661926, "sampling/importance_sampling_ratio/mean": 1.014532446861267, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47727346792817116, "clip_ratio/low_mean": 0.03369492199271917, "clip_ratio/low_min": 0.03369492199271917, "clip_ratio/high_mean": 0.034729897044599056, "clip_ratio/high_max": 0.034729897044599056, "clip_ratio/region_mean": 0.06842481903731823, "reward_total_mean": 0.9344394207000732, "reward_meter_mean": 0.9344394207000732, "reward_meter_std": 0.060838740319013596, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9344394207000732, "reward_total_composite_std": 0.060838740319013596, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 494.0} {"timestamp_utc": "2026-04-11T20:24:22Z", "mode": "train", "global_step": 495, "epoch": 0.01911492122335496, "loss": -0.0075, "grad_norm": 12.706208229064941, "learning_rate": 8.503030303030304e-06, "num_tokens": 1069485.0, "completions/mean_length": 30.875, "completions/min_length": 28.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9881634712219238, "rewards/meter/std": 0.007193571422249079, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9881634712219238, "rewards/total_composite/std": 0.007193571422249079, "reward": 0.9881634712219238, "reward_std": 0.007193575147539377, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.060055576264858246, "sampling/sampling_logp_difference/max": 1.3624719381332397, "sampling/importance_sampling_ratio/min": 0.25602710247039795, "sampling/importance_sampling_ratio/mean": 1.0025434494018555, "sampling/importance_sampling_ratio/max": 1.6685515642166138, "entropy": 0.3115545194596052, "clip_ratio/low_mean": 0.024894393514841795, "clip_ratio/low_min": 0.024894393514841795, "clip_ratio/high_mean": 0.027538669761270285, "clip_ratio/high_max": 0.027538669761270285, "clip_ratio/region_mean": 0.05243306327611208, "reward_total_mean": 0.9881634712219238, "reward_meter_mean": 0.9881634712219238, "reward_meter_std": 0.007193571422249079, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9881634712219238, "reward_total_composite_std": 0.007193571422249079, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 495.0} {"timestamp_utc": "2026-04-11T20:24:27Z", "mode": "train", "global_step": 496, "epoch": 0.019153537225826384, "loss": -0.0266, "grad_norm": 3.555715322494507, "learning_rate": 8.5e-06, "num_tokens": 1071673.0, "completions/mean_length": 92.5, "completions/min_length": 84.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.5, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9900758862495422, "rewards/meter/std": 0.004209411796182394, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9900758862495422, "rewards/total_composite/std": 0.004209411796182394, "reward": 0.9900758862495422, "reward_std": 0.004209410399198532, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02267627976834774, "sampling/sampling_logp_difference/max": 0.8580310344696045, "sampling/importance_sampling_ratio/min": 0.42399612069129944, "sampling/importance_sampling_ratio/mean": 1.0048907995224, "sampling/importance_sampling_ratio/max": 1.8225340843200684, "entropy": 0.17256110534071922, "clip_ratio/low_mean": 0.01094052626285702, "clip_ratio/low_min": 0.01094052626285702, "clip_ratio/high_mean": 0.013113626977428794, "clip_ratio/high_max": 0.013113626977428794, "clip_ratio/region_mean": 0.024054153240285814, "reward_total_mean": 0.9900758862495422, "reward_meter_mean": 0.9900758862495422, "reward_meter_std": 0.004209411796182394, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9900758862495422, "reward_total_composite_std": 0.004209411796182394, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 496.0} {"timestamp_utc": "2026-04-11T20:24:32Z", "mode": "train", "global_step": 497, "epoch": 0.019192153228297808, "loss": -0.0116, "grad_norm": 4.990721225738525, "learning_rate": 8.496969696969697e-06, "num_tokens": 1073467.0, "completions/mean_length": 73.25, "completions/min_length": 67.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9253207445144653, "rewards/meter/std": 0.19340236485004425, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9253207445144653, "rewards/total_composite/std": 0.19340236485004425, "reward": 0.9253207445144653, "reward_std": 0.19340236485004425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08409885317087173, "sampling/sampling_logp_difference/max": 2.713935375213623, "sampling/importance_sampling_ratio/min": 0.06627547740936279, "sampling/importance_sampling_ratio/mean": 1.0199859142303467, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5875852480530739, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/high_mean": 0.055402441415935755, "clip_ratio/high_max": 0.055402441415935755, "clip_ratio/region_mean": 0.06254529859870672, "reward_total_mean": 0.9253207445144653, "reward_meter_mean": 0.9253207445144653, "reward_meter_std": 0.19340236485004425, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9253207445144653, "reward_total_composite_std": 0.19340236485004425, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 497.0} {"timestamp_utc": "2026-04-11T20:24:37Z", "mode": "train", "global_step": 498, "epoch": 0.019230769230769232, "loss": -0.0224, "grad_norm": 3.0449001789093018, "learning_rate": 8.493939393939394e-06, "num_tokens": 1075516.0, "completions/mean_length": 91.125, "completions/min_length": 78.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.125, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9902229905128479, "rewards/meter/std": 0.003602617187425494, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9902229905128479, "rewards/total_composite/std": 0.003602617187425494, "reward": 0.9902229905128479, "reward_std": 0.0036026162561029196, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023890873417258263, "sampling/sampling_logp_difference/max": 1.4790611267089844, "sampling/importance_sampling_ratio/min": 0.2278515249490738, "sampling/importance_sampling_ratio/mean": 0.997484564781189, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1501608146354556, "clip_ratio/low_mean": 0.01110077218618244, "clip_ratio/low_min": 0.01110077218618244, "clip_ratio/high_mean": 0.010509460349567235, "clip_ratio/high_max": 0.010509460349567235, "clip_ratio/region_mean": 0.021610232535749674, "reward_total_mean": 0.9902229905128479, "reward_meter_mean": 0.9902229905128479, "reward_meter_std": 0.003602617187425494, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9902229905128479, "reward_total_composite_std": 0.003602617187425494, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 498.0} {"timestamp_utc": "2026-04-11T20:24:42Z", "mode": "train", "global_step": 499, "epoch": 0.019269385233240656, "loss": 0.0379, "grad_norm": 4.565327167510986, "learning_rate": 8.490909090909092e-06, "num_tokens": 1077319.0, "completions/mean_length": 68.375, "completions/min_length": 58.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.375, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9964392185211182, "rewards/meter/std": 0.0012064595939591527, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9964392185211182, "rewards/total_composite/std": 0.0012064595939591527, "reward": 0.9964392185211182, "reward_std": 0.0012064601760357618, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032802514731884, "sampling/sampling_logp_difference/max": 1.5548906326293945, "sampling/importance_sampling_ratio/min": 0.21121247112751007, "sampling/importance_sampling_ratio/mean": 1.0085853338241577, "sampling/importance_sampling_ratio/max": 1.8629289865493774, "entropy": 0.23555130790919065, "clip_ratio/low_mean": 0.005158253246918321, "clip_ratio/low_min": 0.005158253246918321, "clip_ratio/high_mean": 0.01690437039360404, "clip_ratio/high_max": 0.01690437039360404, "clip_ratio/region_mean": 0.02206262364052236, "reward_total_mean": 0.9964392185211182, "reward_meter_mean": 0.9964392185211182, "reward_meter_std": 0.0012064595939591527, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9964392185211182, "reward_total_composite_std": 0.0012064595939591527, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 499.0} {"timestamp_utc": "2026-04-11T20:24:47Z", "mode": "train", "global_step": 500, "epoch": 0.01930800123571208, "loss": -0.0131, "grad_norm": 4.640279769897461, "learning_rate": 8.487878787878789e-06, "num_tokens": 1079123.0, "completions/mean_length": 77.5, "completions/min_length": 73.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9947643280029297, "rewards/meter/std": 0.0026077541988343, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9947643280029297, "rewards/total_composite/std": 0.0026077541988343, "reward": 0.9947643280029297, "reward_std": 0.0026077530346810818, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05810660123825073, "sampling/sampling_logp_difference/max": 1.3875494003295898, "sampling/importance_sampling_ratio/min": 0.24968643486499786, "sampling/importance_sampling_ratio/mean": 1.0034871101379395, "sampling/importance_sampling_ratio/max": 1.9909405708312988, "entropy": 0.35560205206274986, "clip_ratio/low_mean": 0.01627026009373367, "clip_ratio/low_min": 0.01627026009373367, "clip_ratio/high_mean": 0.03529414697550237, "clip_ratio/high_max": 0.03529414697550237, "clip_ratio/region_mean": 0.05156440706923604, "reward_total_mean": 0.9947643280029297, "reward_meter_mean": 0.9947643280029297, "reward_meter_std": 0.0026077541988343, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9947643280029297, "reward_total_composite_std": 0.0026077541988343, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 500.0} {"timestamp_utc": "2026-04-11T20:26:21Z", "mode": "eval", "global_step": 500, "epoch": 0.01930800123571208, "eval_loss": NaN, "eval_runtime": 93.945, "eval_samples_per_second": 1.107, "eval_steps_per_second": 0.138, "eval_num_tokens": 1079123.0, "eval_completions/mean_length": 282.3942307692308, "eval_completions/min_length": 56.53846153846154, "eval_completions/max_length": 504.3076923076923, "eval_completions/clipped_ratio": 0.27884615384615385, "eval_completions/mean_terminated_length": 189.76557100736179, "eval_completions/min_terminated_length": 56.53846153846154, "eval_completions/max_terminated_length": 360.61538461538464, "eval_rewards/meter/mean": 0.806977744285877, "eval_rewards/meter/std": 0.28674349504021496, "eval_rewards/count_adherence/mean": 0.7540726845081036, "eval_rewards/count_adherence/std": 0.26829315836612994, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.06280488005051246, "eval_rewards/total_composite/mean": 0.5957829631291903, "eval_rewards/total_composite/std": 0.3549887262857877, "eval_reward": 0.5957829631291903, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.02426047422564947, "eval_sampling/sampling_logp_difference/max": 1.013349202963022, "eval_sampling/importance_sampling_ratio/min": 0.38570847190343416, "eval_sampling/importance_sampling_ratio/mean": 1.006858468055725, "eval_sampling/importance_sampling_ratio/max": 1.4029179719778209, "eval_entropy": 0.2863846653356002, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5957829631291903, "eval_reward_meter_mean": 0.806977744285877, "eval_reward_meter_std": 0.28674349504021496, "eval_reward_count_adherence_mean": 0.7540726845081036, "eval_reward_count_adherence_std": 0.26829315836612994, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.06280488005051246, "eval_reward_total_composite_mean": 0.5957829631291903, "eval_reward_total_composite_std": 0.3549887262857877, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 500.0} {"timestamp_utc": "2026-04-11T20:26:32Z", "mode": "train", "global_step": 501, "epoch": 0.019346617238183504, "loss": 0.0695, "grad_norm": 5.740342617034912, "learning_rate": 8.484848484848486e-06, "num_tokens": 1081252.0, "completions/mean_length": 106.125, "completions/min_length": 94.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.125, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.78095543384552, "rewards/meter/std": 0.2571522295475006, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.78095543384552, "rewards/total_composite/std": 0.2571522295475006, "reward": 0.78095543384552, "reward_std": 0.2571522295475006, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0388408899307251, "sampling/sampling_logp_difference/max": 1.936103343963623, "sampling/importance_sampling_ratio/min": 0.14426499605178833, "sampling/importance_sampling_ratio/mean": 1.000687599182129, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22485548444092274, "clip_ratio/low_mean": 0.006404084269888699, "clip_ratio/low_min": 0.006404084269888699, "clip_ratio/high_mean": 0.023434267612174153, "clip_ratio/high_max": 0.023434267612174153, "clip_ratio/region_mean": 0.029838351882062852, "reward_total_mean": 0.78095543384552, "reward_meter_mean": 0.78095543384552, "reward_meter_std": 0.2571522295475006, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.78095543384552, "reward_total_composite_std": 0.2571522295475006, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 501.0} {"timestamp_utc": "2026-04-11T20:26:37Z", "mode": "train", "global_step": 502, "epoch": 0.019385233240654928, "loss": -0.0119, "grad_norm": 5.4612603187561035, "learning_rate": 8.481818181818182e-06, "num_tokens": 1083101.0, "completions/mean_length": 74.125, "completions/min_length": 66.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9958293437957764, "rewards/meter/std": 0.0022998028434813023, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9958293437957764, "rewards/total_composite/std": 0.0022998028434813023, "reward": 0.9958293437957764, "reward_std": 0.0022997905034571886, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07660207152366638, "sampling/sampling_logp_difference/max": 1.3164176940917969, "sampling/importance_sampling_ratio/min": 0.2680939733982086, "sampling/importance_sampling_ratio/mean": 0.9978984594345093, "sampling/importance_sampling_ratio/max": 1.5113362073898315, "entropy": 0.6134384572505951, "clip_ratio/low_mean": 0.02120496891438961, "clip_ratio/low_min": 0.02120496891438961, "clip_ratio/high_mean": 0.04436044883914292, "clip_ratio/high_max": 0.04436044883914292, "clip_ratio/region_mean": 0.06556541775353253, "reward_total_mean": 0.9958293437957764, "reward_meter_mean": 0.9958293437957764, "reward_meter_std": 0.0022998028434813023, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9958293437957764, "reward_total_composite_std": 0.0022998028434813023, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 502.0} {"timestamp_utc": "2026-04-11T20:26:45Z", "mode": "train", "global_step": 503, "epoch": 0.019423849243126352, "loss": 0.0356, "grad_norm": 2.5192019939422607, "learning_rate": 8.478787878787879e-06, "num_tokens": 1086421.0, "completions/mean_length": 224.0, "completions/min_length": 202.0, "completions/max_length": 243.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 224.0, "completions/min_terminated_length": 202.0, "completions/max_terminated_length": 243.0, "rewards/meter/mean": 0.9007662534713745, "rewards/meter/std": 0.15409159660339355, "rewards/count_adherence/mean": 0.7916666269302368, "rewards/count_adherence/std": 0.07715165615081787, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7093769311904907, "rewards/total_composite/std": 0.12365645170211792, "reward": 0.7093769311904907, "reward_std": 0.12365645170211792, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02353961206972599, "sampling/sampling_logp_difference/max": 1.0005009174346924, "sampling/importance_sampling_ratio/min": 0.49309608340263367, "sampling/importance_sampling_ratio/mean": 1.005336880683899, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2083021691069007, "clip_ratio/low_mean": 0.011187375290319324, "clip_ratio/low_min": 0.011187375290319324, "clip_ratio/high_mean": 0.016571022104471922, "clip_ratio/high_max": 0.016571022104471922, "clip_ratio/region_mean": 0.027758397394791245, "reward_total_mean": 0.7093769311904907, "reward_meter_mean": 0.9007662534713745, "reward_meter_std": 0.15409159660339355, "reward_count_adherence_mean": 0.7916666269302368, "reward_count_adherence_std": 0.07715165615081787, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7093769311904907, "reward_total_composite_std": 0.12365645170211792, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 503.0} {"timestamp_utc": "2026-04-11T20:26:50Z", "mode": "train", "global_step": 504, "epoch": 0.019462465245597776, "loss": 0.0237, "grad_norm": 7.049271583557129, "learning_rate": 8.475757575757576e-06, "num_tokens": 1088370.0, "completions/mean_length": 77.625, "completions/min_length": 70.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.625, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.7662866115570068, "rewards/meter/std": 0.42396625876426697, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7662866115570068, "rewards/total_composite/std": 0.42396625876426697, "reward": 0.7662866115570068, "reward_std": 0.4239662289619446, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07983585447072983, "sampling/sampling_logp_difference/max": 1.5671510696411133, "sampling/importance_sampling_ratio/min": 0.2086387276649475, "sampling/importance_sampling_ratio/mean": 1.0156745910644531, "sampling/importance_sampling_ratio/max": 1.7661075592041016, "entropy": 0.6513996087014675, "clip_ratio/low_mean": 0.0136233662487939, "clip_ratio/low_min": 0.0136233662487939, "clip_ratio/high_mean": 0.04974571894854307, "clip_ratio/high_max": 0.04974571894854307, "clip_ratio/region_mean": 0.06336908519733697, "reward_total_mean": 0.7662866115570068, "reward_meter_mean": 0.7662866115570068, "reward_meter_std": 0.42396625876426697, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7662866115570068, "reward_total_composite_std": 0.42396625876426697, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 504.0} {"timestamp_utc": "2026-04-11T20:26:55Z", "mode": "train", "global_step": 505, "epoch": 0.0195010812480692, "loss": 0.0374, "grad_norm": 8.706897735595703, "learning_rate": 8.472727272727274e-06, "num_tokens": 1090079.0, "completions/mean_length": 55.625, "completions/min_length": 48.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.625, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9907145500183105, "rewards/meter/std": 0.0025706093292683363, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9907145500183105, "rewards/total_composite/std": 0.0025706093292683363, "reward": 0.9907145500183105, "reward_std": 0.002570599550381303, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06765672564506531, "sampling/sampling_logp_difference/max": 1.5128650665283203, "sampling/importance_sampling_ratio/min": 0.22027796506881714, "sampling/importance_sampling_ratio/mean": 1.0099107027053833, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5484390016645193, "clip_ratio/low_mean": 0.0246689785271883, "clip_ratio/low_min": 0.0246689785271883, "clip_ratio/high_mean": 0.020820592530071735, "clip_ratio/high_max": 0.020820592530071735, "clip_ratio/region_mean": 0.045489571057260036, "reward_total_mean": 0.9907145500183105, "reward_meter_mean": 0.9907145500183105, "reward_meter_std": 0.0025706093292683363, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9907145500183105, "reward_total_composite_std": 0.0025706093292683363, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 505.0} {"timestamp_utc": "2026-04-11T20:27:00Z", "mode": "train", "global_step": 506, "epoch": 0.019539697250540625, "loss": -0.1105, "grad_norm": 9.833094596862793, "learning_rate": 8.46969696969697e-06, "num_tokens": 1091745.0, "completions/mean_length": 54.25, "completions/min_length": 40.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.25, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8061291575431824, "rewards/meter/std": 0.3550299108028412, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8061291575431824, "rewards/total_composite/std": 0.3550299108028412, "reward": 0.8061291575431824, "reward_std": 0.3550299406051636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08280141651630402, "sampling/sampling_logp_difference/max": 1.5751066207885742, "sampling/importance_sampling_ratio/min": 0.2069854736328125, "sampling/importance_sampling_ratio/mean": 1.0167040824890137, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9682378023862839, "clip_ratio/low_mean": 0.04866071417927742, "clip_ratio/low_min": 0.04866071417927742, "clip_ratio/high_mean": 0.04047576058655977, "clip_ratio/high_max": 0.04047576058655977, "clip_ratio/region_mean": 0.08913647476583719, "reward_total_mean": 0.8061291575431824, "reward_meter_mean": 0.8061291575431824, "reward_meter_std": 0.3550299108028412, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8061291575431824, "reward_total_composite_std": 0.3550299108028412, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 506.0} {"timestamp_utc": "2026-04-11T20:27:05Z", "mode": "train", "global_step": 507, "epoch": 0.01957831325301205, "loss": 0.0027, "grad_norm": 7.7388482093811035, "learning_rate": 8.466666666666668e-06, "num_tokens": 1093764.0, "completions/mean_length": 71.375, "completions/min_length": 58.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9402273893356323, "rewards/meter/std": 0.08104795962572098, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9402273893356323, "rewards/total_composite/std": 0.08104795962572098, "reward": 0.9402273893356323, "reward_std": 0.08104795962572098, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09213840216398239, "sampling/sampling_logp_difference/max": 1.4014253616333008, "sampling/importance_sampling_ratio/min": 0.24624574184417725, "sampling/importance_sampling_ratio/mean": 1.0046333074569702, "sampling/importance_sampling_ratio/max": 1.8226356506347656, "entropy": 0.7916622683405876, "clip_ratio/low_mean": 0.019542983267456293, "clip_ratio/low_min": 0.019542983267456293, "clip_ratio/high_mean": 0.06833233823999763, "clip_ratio/high_max": 0.06833233823999763, "clip_ratio/region_mean": 0.08787532150745392, "reward_total_mean": 0.9402273893356323, "reward_meter_mean": 0.9402273893356323, "reward_meter_std": 0.08104795962572098, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9402273893356323, "reward_total_composite_std": 0.08104795962572098, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 507.0} {"timestamp_utc": "2026-04-11T20:27:12Z", "mode": "train", "global_step": 508, "epoch": 0.019616929255483473, "loss": -0.0493, "grad_norm": 3.3655362129211426, "learning_rate": 8.463636363636364e-06, "num_tokens": 1096436.0, "completions/mean_length": 156.0, "completions/min_length": 118.0, "completions/max_length": 193.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 156.0, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 193.0, "rewards/meter/mean": 0.9674587845802307, "rewards/meter/std": 0.08305719494819641, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9363090991973877, "rewards/total_composite/std": 0.11212779581546783, "reward": 0.9363090991973877, "reward_std": 0.11212778836488724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04139424115419388, "sampling/sampling_logp_difference/max": 1.0351219177246094, "sampling/importance_sampling_ratio/min": 0.3582862913608551, "sampling/importance_sampling_ratio/mean": 1.0092945098876953, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37319300696253777, "clip_ratio/low_mean": 0.015961318742483854, "clip_ratio/low_min": 0.015961318742483854, "clip_ratio/high_mean": 0.02592427283525467, "clip_ratio/high_max": 0.02592427283525467, "clip_ratio/region_mean": 0.041885591577738523, "reward_total_mean": 0.9363090991973877, "reward_meter_mean": 0.9674587845802307, "reward_meter_std": 0.08305719494819641, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9363090991973877, "reward_total_composite_std": 0.11212779581546783, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 508.0} {"timestamp_utc": "2026-04-11T20:27:22Z", "mode": "train", "global_step": 509, "epoch": 0.019655545257954897, "loss": -0.1651, "grad_norm": 1.5039535760879517, "learning_rate": 8.460606060606061e-06, "num_tokens": 1098089.0, "completions/mean_length": 121.625, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 65.85714721679688, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.8553266525268555, "rewards/meter/std": 0.3322550356388092, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.850326418876648, "rewards/total_composite/std": 0.3462829291820526, "reward": 0.850326418876648, "reward_std": 0.3462829291820526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12041088938713074, "sampling/sampling_logp_difference/max": 1.2919178009033203, "sampling/importance_sampling_ratio/min": 0.27474337816238403, "sampling/importance_sampling_ratio/mean": 1.0222753286361694, "sampling/importance_sampling_ratio/max": 1.746443271636963, "entropy": 1.1623116582632065, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09690898563712835, "clip_ratio/high_max": 0.09690898563712835, "clip_ratio/region_mean": 0.09690898563712835, "reward_total_mean": 0.850326418876648, "reward_meter_mean": 0.8553266525268555, "reward_meter_std": 0.3322550356388092, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.850326418876648, "reward_total_composite_std": 0.3462829291820526, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 509.0} {"timestamp_utc": "2026-04-11T20:27:32Z", "mode": "train", "global_step": 510, "epoch": 0.01969416126042632, "loss": 0.0177, "grad_norm": 1.3078835010528564, "learning_rate": 8.457575757575758e-06, "num_tokens": 1103402.0, "completions/mean_length": 447.125, "completions/min_length": 398.0, "completions/max_length": 505.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 447.125, "completions/min_terminated_length": 398.0, "completions/max_terminated_length": 505.0, "rewards/meter/mean": 0.7473684549331665, "rewards/meter/std": 0.4575249254703522, "rewards/count_adherence/mean": 0.4464285969734192, "rewards/count_adherence/std": 0.11921756714582443, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.3376147150993347, "rewards/total_composite/std": 0.22579079866409302, "reward": 0.3376147150993347, "reward_std": 0.22579078376293182, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015463379211723804, "sampling/sampling_logp_difference/max": 1.221771240234375, "sampling/importance_sampling_ratio/min": 0.294707715511322, "sampling/importance_sampling_ratio/mean": 1.0023648738861084, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1255875793285668, "clip_ratio/low_mean": 0.00692010746570304, "clip_ratio/low_min": 0.00692010746570304, "clip_ratio/high_mean": 0.006500857998616993, "clip_ratio/high_max": 0.006500857998616993, "clip_ratio/region_mean": 0.013420965464320034, "reward_total_mean": 0.3376147150993347, "reward_meter_mean": 0.7473684549331665, "reward_meter_std": 0.4575249254703522, "reward_count_adherence_mean": 0.4464285969734192, "reward_count_adherence_std": 0.11921756714582443, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.3376147150993347, "reward_total_composite_std": 0.22579079866409302, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 510.0} {"timestamp_utc": "2026-04-11T20:27:37Z", "mode": "train", "global_step": 511, "epoch": 0.019732777262897745, "loss": -0.019, "grad_norm": 10.033199310302734, "learning_rate": 8.454545454545455e-06, "num_tokens": 1105240.0, "completions/mean_length": 63.75, "completions/min_length": 53.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.75, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.6957893371582031, "rewards/meter/std": 0.35945481061935425, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6394338011741638, "rewards/total_composite/std": 0.35790061950683594, "reward": 0.6394338011741638, "reward_std": 0.35790061950683594, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11444373428821564, "sampling/sampling_logp_difference/max": 2.7123990058898926, "sampling/importance_sampling_ratio/min": 0.06637737900018692, "sampling/importance_sampling_ratio/mean": 1.0079913139343262, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0178688187152147, "clip_ratio/low_mean": 0.03445858974009752, "clip_ratio/low_min": 0.03445858974009752, "clip_ratio/high_mean": 0.058790234150364995, "clip_ratio/high_max": 0.058790234150364995, "clip_ratio/region_mean": 0.09324882389046252, "reward_total_mean": 0.6394338011741638, "reward_meter_mean": 0.6957893371582031, "reward_meter_std": 0.35945481061935425, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6394338011741638, "reward_total_composite_std": 0.35790061950683594, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 511.0} {"timestamp_utc": "2026-04-11T20:27:44Z", "mode": "train", "global_step": 512, "epoch": 0.01977139326536917, "loss": -0.0532, "grad_norm": 1.5483953952789307, "learning_rate": 8.451515151515151e-06, "num_tokens": 1108518.0, "completions/mean_length": 225.75, "completions/min_length": 186.0, "completions/max_length": 278.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 225.75, "completions/min_terminated_length": 186.0, "completions/max_terminated_length": 278.0, "rewards/meter/mean": 0.8709026575088501, "rewards/meter/std": 0.35162675380706787, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8210574388504028, "rewards/total_composite/std": 0.34322604537010193, "reward": 0.8210574388504028, "reward_std": 0.34322601556777954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02275705337524414, "sampling/sampling_logp_difference/max": 0.9215679168701172, "sampling/importance_sampling_ratio/min": 0.3978947103023529, "sampling/importance_sampling_ratio/mean": 1.0026448965072632, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19403257127851248, "clip_ratio/low_mean": 0.00832630880177021, "clip_ratio/low_min": 0.00832630880177021, "clip_ratio/high_mean": 0.014946788433007896, "clip_ratio/high_max": 0.014946788433007896, "clip_ratio/region_mean": 0.023273097234778106, "reward_total_mean": 0.8210574388504028, "reward_meter_mean": 0.8709026575088501, "reward_meter_std": 0.35162675380706787, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8210574388504028, "reward_total_composite_std": 0.34322604537010193, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 512.0} {"timestamp_utc": "2026-04-11T20:27:50Z", "mode": "train", "global_step": 513, "epoch": 0.019810009267840593, "loss": -0.0103, "grad_norm": 2.7554123401641846, "learning_rate": 8.44848484848485e-06, "num_tokens": 1110976.0, "completions/mean_length": 149.25, "completions/min_length": 132.0, "completions/max_length": 175.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.25, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 175.0, "rewards/meter/mean": 0.8965615034103394, "rewards/meter/std": 0.14327730238437653, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8653509020805359, "rewards/total_composite/std": 0.14502419531345367, "reward": 0.8653509020805359, "reward_std": 0.14502419531345367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02919297106564045, "sampling/sampling_logp_difference/max": 1.2159080505371094, "sampling/importance_sampling_ratio/min": 0.2964406907558441, "sampling/importance_sampling_ratio/mean": 1.002771258354187, "sampling/importance_sampling_ratio/max": 1.710719108581543, "entropy": 0.19797090534120798, "clip_ratio/low_mean": 0.009132357081398368, "clip_ratio/low_min": 0.009132357081398368, "clip_ratio/high_mean": 0.02126406622119248, "clip_ratio/high_max": 0.02126406622119248, "clip_ratio/region_mean": 0.030396423302590847, "reward_total_mean": 0.8653509020805359, "reward_meter_mean": 0.8965615034103394, "reward_meter_std": 0.14327730238437653, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8653509020805359, "reward_total_composite_std": 0.14502419531345367, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 513.0} {"timestamp_utc": "2026-04-11T20:27:55Z", "mode": "train", "global_step": 514, "epoch": 0.019848625270312018, "loss": 0.0446, "grad_norm": 8.517362594604492, "learning_rate": 8.445454545454547e-06, "num_tokens": 1112726.0, "completions/mean_length": 67.75, "completions/min_length": 60.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.8835674524307251, "rewards/meter/std": 0.292568564414978, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8835674524307251, "rewards/total_composite/std": 0.292568564414978, "reward": 0.8835674524307251, "reward_std": 0.2925685942173004, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06897350400686264, "sampling/sampling_logp_difference/max": 1.4508752822875977, "sampling/importance_sampling_ratio/min": 0.2343650758266449, "sampling/importance_sampling_ratio/mean": 1.009161114692688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40955257788300514, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.051333898678421974, "clip_ratio/high_max": 0.051333898678421974, "clip_ratio/region_mean": 0.054712277138605714, "reward_total_mean": 0.8835674524307251, "reward_meter_mean": 0.8835674524307251, "reward_meter_std": 0.292568564414978, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8835674524307251, "reward_total_composite_std": 0.292568564414978, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 514.0} {"timestamp_utc": "2026-04-11T20:28:00Z", "mode": "train", "global_step": 515, "epoch": 0.01988724127278344, "loss": -0.0222, "grad_norm": 5.492783069610596, "learning_rate": 8.442424242424243e-06, "num_tokens": 1114499.0, "completions/mean_length": 61.625, "completions/min_length": 57.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7765995264053345, "rewards/meter/std": 0.4026697874069214, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7765995264053345, "rewards/total_composite/std": 0.4026697874069214, "reward": 0.7765995264053345, "reward_std": 0.4026697874069214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06085870414972305, "sampling/sampling_logp_difference/max": 1.8261196613311768, "sampling/importance_sampling_ratio/min": 0.2164708971977234, "sampling/importance_sampling_ratio/mean": 1.0022004842758179, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4629223048686981, "clip_ratio/low_mean": 0.021710526663810015, "clip_ratio/low_min": 0.021710526663810015, "clip_ratio/high_mean": 0.043231920688413084, "clip_ratio/high_max": 0.043231920688413084, "clip_ratio/region_mean": 0.0649424473522231, "reward_total_mean": 0.7765995264053345, "reward_meter_mean": 0.7765995264053345, "reward_meter_std": 0.4026697874069214, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7765995264053345, "reward_total_composite_std": 0.4026697874069214, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 515.0} {"timestamp_utc": "2026-04-11T20:28:05Z", "mode": "train", "global_step": 516, "epoch": 0.019925857275254866, "loss": -0.0002, "grad_norm": 13.839630126953125, "learning_rate": 8.43939393939394e-06, "num_tokens": 1116008.0, "completions/mean_length": 32.625, "completions/min_length": 23.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.7724308371543884, "rewards/meter/std": 0.31887128949165344, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7724308371543884, "rewards/total_composite/std": 0.31887128949165344, "reward": 0.7724308371543884, "reward_std": 0.31887128949165344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13659369945526123, "sampling/sampling_logp_difference/max": 1.1804313659667969, "sampling/importance_sampling_ratio/min": 0.30714622139930725, "sampling/importance_sampling_ratio/mean": 1.0144718885421753, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2552415579557419, "clip_ratio/low_mean": 0.04520996939390898, "clip_ratio/low_min": 0.04520996939390898, "clip_ratio/high_mean": 0.07153899129480124, "clip_ratio/high_max": 0.07153899129480124, "clip_ratio/region_mean": 0.11674896068871021, "reward_total_mean": 0.7724308371543884, "reward_meter_mean": 0.7724308371543884, "reward_meter_std": 0.31887128949165344, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7724308371543884, "reward_total_composite_std": 0.31887128949165344, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 516.0} {"timestamp_utc": "2026-04-11T20:28:10Z", "mode": "train", "global_step": 517, "epoch": 0.01996447327772629, "loss": -0.0974, "grad_norm": 17.498676300048828, "learning_rate": 8.436363636363637e-06, "num_tokens": 1117535.0, "completions/mean_length": 24.875, "completions/min_length": 18.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.875, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9236002564430237, "rewards/meter/std": 0.16382306814193726, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9236002564430237, "rewards/total_composite/std": 0.16382306814193726, "reward": 0.9236002564430237, "reward_std": 0.16382306814193726, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13017626106739044, "sampling/sampling_logp_difference/max": 1.2851448059082031, "sampling/importance_sampling_ratio/min": 0.27661052346229553, "sampling/importance_sampling_ratio/mean": 1.0398297309875488, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.112723309546709, "clip_ratio/low_mean": 0.02741228137165308, "clip_ratio/low_min": 0.02741228137165308, "clip_ratio/high_mean": 0.0776620376855135, "clip_ratio/high_max": 0.0776620376855135, "clip_ratio/region_mean": 0.10507431905716658, "reward_total_mean": 0.9236002564430237, "reward_meter_mean": 0.9236002564430237, "reward_meter_std": 0.16382306814193726, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9236002564430237, "reward_total_composite_std": 0.16382306814193726, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 517.0} {"timestamp_utc": "2026-04-11T20:28:15Z", "mode": "train", "global_step": 518, "epoch": 0.020003089280197714, "loss": 0.047, "grad_norm": 15.10335922241211, "learning_rate": 8.433333333333334e-06, "num_tokens": 1119517.0, "completions/mean_length": 61.75, "completions/min_length": 56.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9054743051528931, "rewards/meter/std": 0.21583297848701477, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9054743051528931, "rewards/total_composite/std": 0.21583297848701477, "reward": 0.9054743051528931, "reward_std": 0.21583294868469238, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07659897208213806, "sampling/sampling_logp_difference/max": 1.6656980514526367, "sampling/importance_sampling_ratio/min": 0.18905863165855408, "sampling/importance_sampling_ratio/mean": 1.0084285736083984, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3643992841243744, "clip_ratio/low_mean": 0.007462686393409967, "clip_ratio/low_min": 0.007462686393409967, "clip_ratio/high_mean": 0.049344516824930906, "clip_ratio/high_max": 0.049344516824930906, "clip_ratio/region_mean": 0.056807203218340874, "reward_total_mean": 0.9054743051528931, "reward_meter_mean": 0.9054743051528931, "reward_meter_std": 0.21583297848701477, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9054743051528931, "reward_total_composite_std": 0.21583297848701477, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 518.0} {"timestamp_utc": "2026-04-11T20:28:20Z", "mode": "train", "global_step": 519, "epoch": 0.020041705282669138, "loss": 0.0138, "grad_norm": 8.149646759033203, "learning_rate": 8.43030303030303e-06, "num_tokens": 1121333.0, "completions/mean_length": 70.0, "completions/min_length": 63.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9806559085845947, "rewards/meter/std": 0.02947128564119339, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9806559085845947, "rewards/total_composite/std": 0.02947128564119339, "reward": 0.9806559085845947, "reward_std": 0.029471300542354584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07023530453443527, "sampling/sampling_logp_difference/max": 4.686591625213623, "sampling/importance_sampling_ratio/min": 0.009218051098287106, "sampling/importance_sampling_ratio/mean": 1.0027549266815186, "sampling/importance_sampling_ratio/max": 1.8948339223861694, "entropy": 0.3383452221751213, "clip_ratio/low_mean": 0.007263681618496776, "clip_ratio/low_min": 0.007263681618496776, "clip_ratio/high_mean": 0.04268710082396865, "clip_ratio/high_max": 0.04268710082396865, "clip_ratio/region_mean": 0.049950782442465425, "reward_total_mean": 0.9806559085845947, "reward_meter_mean": 0.9806559085845947, "reward_meter_std": 0.02947128564119339, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9806559085845947, "reward_total_composite_std": 0.02947128564119339, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 519.0} {"timestamp_utc": "2026-04-11T20:28:24Z", "mode": "train", "global_step": 520, "epoch": 0.020080321285140562, "loss": 0.0682, "grad_norm": 8.837297439575195, "learning_rate": 8.427272727272729e-06, "num_tokens": 1123139.0, "completions/mean_length": 55.75, "completions/min_length": 51.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.75, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.8729170560836792, "rewards/meter/std": 0.29740720987319946, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8729170560836792, "rewards/total_composite/std": 0.29740720987319946, "reward": 0.8729170560836792, "reward_std": 0.2974071800708771, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07541073858737946, "sampling/sampling_logp_difference/max": 1.265242338180542, "sampling/importance_sampling_ratio/min": 0.2821709215641022, "sampling/importance_sampling_ratio/mean": 0.9940183162689209, "sampling/importance_sampling_ratio/max": 1.8521945476531982, "entropy": 0.4810672476887703, "clip_ratio/low_mean": 0.007692307699471712, "clip_ratio/low_min": 0.007692307699471712, "clip_ratio/high_mean": 0.07098443387076259, "clip_ratio/high_max": 0.07098443387076259, "clip_ratio/region_mean": 0.0786767415702343, "reward_total_mean": 0.8729170560836792, "reward_meter_mean": 0.8729170560836792, "reward_meter_std": 0.29740720987319946, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8729170560836792, "reward_total_composite_std": 0.29740720987319946, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 520.0} {"timestamp_utc": "2026-04-11T20:28:29Z", "mode": "train", "global_step": 521, "epoch": 0.020118937287611986, "loss": 0.0135, "grad_norm": 16.822296142578125, "learning_rate": 8.424242424242425e-06, "num_tokens": 1124881.0, "completions/mean_length": 59.75, "completions/min_length": 52.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9238840341567993, "rewards/meter/std": 0.19443634152412415, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9238840341567993, "rewards/total_composite/std": 0.19443634152412415, "reward": 0.9238840341567993, "reward_std": 0.19443634152412415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039955347776412964, "sampling/sampling_logp_difference/max": 3.09586501121521, "sampling/importance_sampling_ratio/min": 0.04523586481809616, "sampling/importance_sampling_ratio/mean": 0.9970530867576599, "sampling/importance_sampling_ratio/max": 1.7410789728164673, "entropy": 0.1859533153474331, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/high_mean": 0.022381525253877044, "clip_ratio/high_max": 0.022381525253877044, "clip_ratio/region_mean": 0.03071485902182758, "reward_total_mean": 0.9238840341567993, "reward_meter_mean": 0.9238840341567993, "reward_meter_std": 0.19443634152412415, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9238840341567993, "reward_total_composite_std": 0.19443634152412415, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 521.0} {"timestamp_utc": "2026-04-11T20:28:34Z", "mode": "train", "global_step": 522, "epoch": 0.02015755329008341, "loss": 0.0201, "grad_norm": 5.966022491455078, "learning_rate": 8.421212121212122e-06, "num_tokens": 1126658.0, "completions/mean_length": 75.125, "completions/min_length": 70.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.125, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9940762519836426, "rewards/meter/std": 0.0017088382737711072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9940762519836426, "rewards/total_composite/std": 0.0017088382737711072, "reward": 0.9940762519836426, "reward_std": 0.0017088415334001184, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05746918171644211, "sampling/sampling_logp_difference/max": 1.0010395050048828, "sampling/importance_sampling_ratio/min": 0.3674972355365753, "sampling/importance_sampling_ratio/mean": 1.0083460807800293, "sampling/importance_sampling_ratio/max": 1.9666850566864014, "entropy": 0.3937871027737856, "clip_ratio/low_mean": 0.01347197126597166, "clip_ratio/low_min": 0.01347197126597166, "clip_ratio/high_mean": 0.028243856504559517, "clip_ratio/high_max": 0.028243856504559517, "clip_ratio/region_mean": 0.04171582777053118, "reward_total_mean": 0.9940762519836426, "reward_meter_mean": 0.9940762519836426, "reward_meter_std": 0.0017088382737711072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9940762519836426, "reward_total_composite_std": 0.0017088382737711072, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 522.0} {"timestamp_utc": "2026-04-11T20:28:39Z", "mode": "train", "global_step": 523, "epoch": 0.020196169292554834, "loss": -0.0272, "grad_norm": 4.775942802429199, "learning_rate": 8.418181818181819e-06, "num_tokens": 1128392.0, "completions/mean_length": 62.75, "completions/min_length": 54.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8802205324172974, "rewards/meter/std": 0.31313174962997437, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8802205324172974, "rewards/total_composite/std": 0.31313174962997437, "reward": 0.8802205324172974, "reward_std": 0.31313174962997437, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040101900696754456, "sampling/sampling_logp_difference/max": 1.4806737899780273, "sampling/importance_sampling_ratio/min": 0.22748437523841858, "sampling/importance_sampling_ratio/mean": 0.9997758269309998, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23601900041103363, "clip_ratio/low_mean": 0.004464285913854837, "clip_ratio/low_min": 0.004464285913854837, "clip_ratio/high_mean": 0.04396977473516017, "clip_ratio/high_max": 0.04396977473516017, "clip_ratio/region_mean": 0.04843406064901501, "reward_total_mean": 0.8802205324172974, "reward_meter_mean": 0.8802205324172974, "reward_meter_std": 0.31313174962997437, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8802205324172974, "reward_total_composite_std": 0.31313174962997437, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 523.0} {"timestamp_utc": "2026-04-11T20:28:44Z", "mode": "train", "global_step": 524, "epoch": 0.02023478529502626, "loss": -0.0291, "grad_norm": 8.702469825744629, "learning_rate": 8.415151515151516e-06, "num_tokens": 1130217.0, "completions/mean_length": 62.125, "completions/min_length": 43.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8824061155319214, "rewards/meter/std": 0.2138787806034088, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8824061155319214, "rewards/total_composite/std": 0.2138787806034088, "reward": 0.8824061155319214, "reward_std": 0.2138787806034088, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10073237121105194, "sampling/sampling_logp_difference/max": 1.5494084358215332, "sampling/importance_sampling_ratio/min": 0.21237356960773468, "sampling/importance_sampling_ratio/mean": 1.0184638500213623, "sampling/importance_sampling_ratio/max": 1.9018948078155518, "entropy": 1.2009354755282402, "clip_ratio/low_mean": 0.0262486576102674, "clip_ratio/low_min": 0.0262486576102674, "clip_ratio/high_mean": 0.0902106543071568, "clip_ratio/high_max": 0.0902106543071568, "clip_ratio/region_mean": 0.1164593119174242, "reward_total_mean": 0.8824061155319214, "reward_meter_mean": 0.8824061155319214, "reward_meter_std": 0.2138787806034088, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8824061155319214, "reward_total_composite_std": 0.2138787806034088, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 524.0} {"timestamp_utc": "2026-04-11T20:28:49Z", "mode": "train", "global_step": 525, "epoch": 0.020273401297497683, "loss": 0.0389, "grad_norm": 6.02722692489624, "learning_rate": 8.412121212121212e-06, "num_tokens": 1131849.0, "completions/mean_length": 59.0, "completions/min_length": 53.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9700571298599243, "rewards/meter/std": 0.05985637754201889, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9700571298599243, "rewards/total_composite/std": 0.05985637754201889, "reward": 0.9700571298599243, "reward_std": 0.059856388717889786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05891498550772667, "sampling/sampling_logp_difference/max": 1.0405960083007812, "sampling/importance_sampling_ratio/min": 0.3532440662384033, "sampling/importance_sampling_ratio/mean": 1.01723313331604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3998655714094639, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/high_mean": 0.027191198198124766, "clip_ratio/high_max": 0.027191198198124766, "clip_ratio/region_mean": 0.031037352047860622, "reward_total_mean": 0.9700571298599243, "reward_meter_mean": 0.9700571298599243, "reward_meter_std": 0.05985637754201889, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9700571298599243, "reward_total_composite_std": 0.05985637754201889, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 525.0} {"timestamp_utc": "2026-04-11T20:28:53Z", "mode": "train", "global_step": 526, "epoch": 0.020312017299969107, "loss": 0.0231, "grad_norm": 5.217512607574463, "learning_rate": 8.40909090909091e-06, "num_tokens": 1133805.0, "completions/mean_length": 70.5, "completions/min_length": 64.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9931546449661255, "rewards/meter/std": 0.004212968982756138, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9931546449661255, "rewards/total_composite/std": 0.004212968982756138, "reward": 0.9931546449661255, "reward_std": 0.00421295827254653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0634373277425766, "sampling/sampling_logp_difference/max": 1.5378689765930176, "sampling/importance_sampling_ratio/min": 0.214838445186615, "sampling/importance_sampling_ratio/mean": 1.0042046308517456, "sampling/importance_sampling_ratio/max": 1.9715811014175415, "entropy": 0.4445030000060797, "clip_ratio/low_mean": 0.021122918464243412, "clip_ratio/low_min": 0.021122918464243412, "clip_ratio/high_mean": 0.029155581374652684, "clip_ratio/high_max": 0.029155581374652684, "clip_ratio/region_mean": 0.050278499838896096, "reward_total_mean": 0.9931546449661255, "reward_meter_mean": 0.9931546449661255, "reward_meter_std": 0.004212968982756138, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9931546449661255, "reward_total_composite_std": 0.004212968982756138, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 526.0} {"timestamp_utc": "2026-04-11T20:28:58Z", "mode": "train", "global_step": 527, "epoch": 0.02035063330244053, "loss": 0.006, "grad_norm": 2.7269952297210693, "learning_rate": 8.406060606060606e-06, "num_tokens": 1135576.0, "completions/mean_length": 75.375, "completions/min_length": 64.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9965004920959473, "rewards/meter/std": 0.0009321427205577493, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9965004920959473, "rewards/total_composite/std": 0.0009321427205577493, "reward": 0.9965004920959473, "reward_std": 0.000932130089495331, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04060669243335724, "sampling/sampling_logp_difference/max": 0.822037935256958, "sampling/importance_sampling_ratio/min": 0.4395350217819214, "sampling/importance_sampling_ratio/mean": 1.0171750783920288, "sampling/importance_sampling_ratio/max": 1.7970446348190308, "entropy": 0.3602394051849842, "clip_ratio/low_mean": 0.01564345066435635, "clip_ratio/low_min": 0.01564345066435635, "clip_ratio/high_mean": 0.016355944564566016, "clip_ratio/high_max": 0.016355944564566016, "clip_ratio/region_mean": 0.03199939522892237, "reward_total_mean": 0.9965004920959473, "reward_meter_mean": 0.9965004920959473, "reward_meter_std": 0.0009321427205577493, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9965004920959473, "reward_total_composite_std": 0.0009321427205577493, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 527.0} {"timestamp_utc": "2026-04-11T20:29:03Z", "mode": "train", "global_step": 528, "epoch": 0.020389249304911955, "loss": 0.0118, "grad_norm": 9.207352638244629, "learning_rate": 8.403030303030304e-06, "num_tokens": 1137326.0, "completions/mean_length": 66.75, "completions/min_length": 56.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9670517444610596, "rewards/meter/std": 0.06938152760267258, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9670517444610596, "rewards/total_composite/std": 0.06938152760267258, "reward": 0.9670517444610596, "reward_std": 0.06938153505325317, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08710537105798721, "sampling/sampling_logp_difference/max": 1.324056625366211, "sampling/importance_sampling_ratio/min": 0.26605382561683655, "sampling/importance_sampling_ratio/mean": 0.9962921738624573, "sampling/importance_sampling_ratio/max": 1.738789677619934, "entropy": 0.784766998142004, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.07623076229356229, "clip_ratio/high_max": 0.07623076229356229, "clip_ratio/region_mean": 0.08001864119432867, "reward_total_mean": 0.9670517444610596, "reward_meter_mean": 0.9670517444610596, "reward_meter_std": 0.06938152760267258, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9670517444610596, "reward_total_composite_std": 0.06938152760267258, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 528.0} {"timestamp_utc": "2026-04-11T20:29:08Z", "mode": "train", "global_step": 529, "epoch": 0.02042786530738338, "loss": -0.0752, "grad_norm": 7.656765460968018, "learning_rate": 8.400000000000001e-06, "num_tokens": 1139126.0, "completions/mean_length": 61.0, "completions/min_length": 47.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9007197618484497, "rewards/meter/std": 0.2582366466522217, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9007197618484497, "rewards/total_composite/std": 0.2582366466522217, "reward": 0.9007197618484497, "reward_std": 0.2582366466522217, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03346162289381027, "sampling/sampling_logp_difference/max": 1.2477359771728516, "sampling/importance_sampling_ratio/min": 0.2871541976928711, "sampling/importance_sampling_ratio/mean": 1.0085376501083374, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30135350674390793, "clip_ratio/low_mean": 0.029255319386720657, "clip_ratio/low_min": 0.029255319386720657, "clip_ratio/high_mean": 0.015567422262392938, "clip_ratio/high_max": 0.015567422262392938, "clip_ratio/region_mean": 0.044822741649113595, "reward_total_mean": 0.9007197618484497, "reward_meter_mean": 0.9007197618484497, "reward_meter_std": 0.2582366466522217, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9007197618484497, "reward_total_composite_std": 0.2582366466522217, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 529.0} {"timestamp_utc": "2026-04-11T20:29:19Z", "mode": "train", "global_step": 530, "epoch": 0.020466481309854803, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 8.396969696969698e-06, "num_tokens": 1140846.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9937265515327454, "rewards/meter/std": 0.002974079456180334, "rewards/count_adherence/mean": 0.7941176891326904, "rewards/count_adherence/std": 0.03144249692559242, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7890878915786743, "rewards/total_composite/std": 0.029924675822257996, "reward": 0.7890878915786743, "reward_std": 0.029924683272838593, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7890878915786743, "reward_meter_mean": 0.9937265515327454, "reward_meter_std": 0.002974079456180334, "reward_count_adherence_mean": 0.7941176891326904, "reward_count_adherence_std": 0.03144249692559242, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7890878915786743, "reward_total_composite_std": 0.029924675822257996, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 530.0} {"timestamp_utc": "2026-04-11T20:29:23Z", "mode": "train", "global_step": 531, "epoch": 0.020505097312326227, "loss": 0.0262, "grad_norm": 15.459403038024902, "learning_rate": 8.393939393939394e-06, "num_tokens": 1142401.0, "completions/mean_length": 44.375, "completions/min_length": 32.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.564997673034668, "rewards/meter/std": 0.3845665752887726, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.564997673034668, "rewards/total_composite/std": 0.3845665752887726, "reward": 0.564997673034668, "reward_std": 0.3845665454864502, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17102597653865814, "sampling/sampling_logp_difference/max": 2.347954750061035, "sampling/importance_sampling_ratio/min": 0.09556441009044647, "sampling/importance_sampling_ratio/mean": 1.0244461297988892, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1344088204205036, "clip_ratio/low_mean": 0.08407607581466436, "clip_ratio/low_min": 0.08407607581466436, "clip_ratio/high_mean": 0.0648979377001524, "clip_ratio/high_max": 0.0648979377001524, "clip_ratio/region_mean": 0.14897401351481676, "reward_total_mean": 0.564997673034668, "reward_meter_mean": 0.564997673034668, "reward_meter_std": 0.3845665752887726, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.564997673034668, "reward_total_composite_std": 0.3845665752887726, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 531.0} {"timestamp_utc": "2026-04-11T20:29:28Z", "mode": "train", "global_step": 532, "epoch": 0.02054371331479765, "loss": 0.0452, "grad_norm": 11.56308364868164, "learning_rate": 8.390909090909091e-06, "num_tokens": 1144260.0, "completions/mean_length": 60.375, "completions/min_length": 56.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.772533655166626, "rewards/meter/std": 0.38782942295074463, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.772533655166626, "rewards/total_composite/std": 0.38782942295074463, "reward": 0.772533655166626, "reward_std": 0.38782939314842224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08298543095588684, "sampling/sampling_logp_difference/max": 1.7821917533874512, "sampling/importance_sampling_ratio/min": 0.16826894879341125, "sampling/importance_sampling_ratio/mean": 1.002143383026123, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5591697357594967, "clip_ratio/low_mean": 0.01616688398644328, "clip_ratio/low_min": 0.01616688398644328, "clip_ratio/high_mean": 0.0692659798078239, "clip_ratio/high_max": 0.0692659798078239, "clip_ratio/region_mean": 0.08543286379426718, "reward_total_mean": 0.772533655166626, "reward_meter_mean": 0.772533655166626, "reward_meter_std": 0.38782942295074463, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.772533655166626, "reward_total_composite_std": 0.38782942295074463, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 532.0} {"timestamp_utc": "2026-04-11T20:29:33Z", "mode": "train", "global_step": 533, "epoch": 0.020582329317269075, "loss": -0.0447, "grad_norm": 7.510979652404785, "learning_rate": 8.387878787878788e-06, "num_tokens": 1145610.0, "completions/mean_length": 31.75, "completions/min_length": 22.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.75, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.994328498840332, "rewards/meter/std": 0.00353585509583354, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.994328498840332, "rewards/total_composite/std": 0.00353585509583354, "reward": 0.994328498840332, "reward_std": 0.00353585509583354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07902076840400696, "sampling/sampling_logp_difference/max": 1.1082839965820312, "sampling/importance_sampling_ratio/min": 0.33012497425079346, "sampling/importance_sampling_ratio/mean": 0.9933980703353882, "sampling/importance_sampling_ratio/max": 1.9707249402999878, "entropy": 0.6379514243453741, "clip_ratio/low_mean": 0.02431722730398178, "clip_ratio/low_min": 0.02431722730398178, "clip_ratio/high_mean": 0.059228118509054184, "clip_ratio/high_max": 0.059228118509054184, "clip_ratio/region_mean": 0.08354534581303596, "reward_total_mean": 0.994328498840332, "reward_meter_mean": 0.994328498840332, "reward_meter_std": 0.00353585509583354, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.994328498840332, "reward_total_composite_std": 0.00353585509583354, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 533.0} {"timestamp_utc": "2026-04-11T20:29:38Z", "mode": "train", "global_step": 534, "epoch": 0.0206209453197405, "loss": 0.0332, "grad_norm": 4.608026504516602, "learning_rate": 8.384848484848485e-06, "num_tokens": 1147340.0, "completions/mean_length": 56.25, "completions/min_length": 49.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9940440058708191, "rewards/meter/std": 0.0018207214307039976, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9940440058708191, "rewards/total_composite/std": 0.0018207214307039976, "reward": 0.9940440058708191, "reward_std": 0.0018207177054136992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02367156185209751, "sampling/sampling_logp_difference/max": 0.9618037343025208, "sampling/importance_sampling_ratio/min": 0.3822028934955597, "sampling/importance_sampling_ratio/mean": 1.0104416608810425, "sampling/importance_sampling_ratio/max": 1.6200429201126099, "entropy": 0.17711176443845034, "clip_ratio/low_mean": 0.008662280859425664, "clip_ratio/low_min": 0.008662280859425664, "clip_ratio/high_mean": 0.00898129097186029, "clip_ratio/high_max": 0.00898129097186029, "clip_ratio/region_mean": 0.017643571831285954, "reward_total_mean": 0.9940440058708191, "reward_meter_mean": 0.9940440058708191, "reward_meter_std": 0.0018207214307039976, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9940440058708191, "reward_total_composite_std": 0.0018207214307039976, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 534.0} {"timestamp_utc": "2026-04-11T20:29:43Z", "mode": "train", "global_step": 535, "epoch": 0.020659561322211924, "loss": 0.0063, "grad_norm": 5.5635881423950195, "learning_rate": 8.381818181818183e-06, "num_tokens": 1149134.0, "completions/mean_length": 66.25, "completions/min_length": 58.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.7755165100097656, "rewards/meter/std": 0.3405035734176636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7755165100097656, "rewards/total_composite/std": 0.3405035734176636, "reward": 0.7755165100097656, "reward_std": 0.3405035436153412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.044020526111125946, "sampling/sampling_logp_difference/max": 1.8177876472473145, "sampling/importance_sampling_ratio/min": 0.16238459944725037, "sampling/importance_sampling_ratio/mean": 1.0043402910232544, "sampling/importance_sampling_ratio/max": 1.6945619583129883, "entropy": 0.3021476771682501, "clip_ratio/low_mean": 0.008223684038966894, "clip_ratio/low_min": 0.008223684038966894, "clip_ratio/high_mean": 0.02654118835926056, "clip_ratio/high_max": 0.02654118835926056, "clip_ratio/region_mean": 0.03476487239822745, "reward_total_mean": 0.7755165100097656, "reward_meter_mean": 0.7755165100097656, "reward_meter_std": 0.3405035734176636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7755165100097656, "reward_total_composite_std": 0.3405035734176636, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 535.0} {"timestamp_utc": "2026-04-11T20:29:48Z", "mode": "train", "global_step": 536, "epoch": 0.020698177324683348, "loss": -0.0758, "grad_norm": 8.588635444641113, "learning_rate": 8.37878787878788e-06, "num_tokens": 1150740.0, "completions/mean_length": 53.75, "completions/min_length": 32.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7046811580657959, "rewards/meter/std": 0.4017917513847351, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7046811580657959, "rewards/total_composite/std": 0.4017917513847351, "reward": 0.7046811580657959, "reward_std": 0.4017917513847351, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10004501044750214, "sampling/sampling_logp_difference/max": 1.1618900299072266, "sampling/importance_sampling_ratio/min": 0.31289422512054443, "sampling/importance_sampling_ratio/mean": 1.0192784070968628, "sampling/importance_sampling_ratio/max": 1.8760080337524414, "entropy": 1.1170207932591438, "clip_ratio/low_mean": 0.043238147161901, "clip_ratio/low_min": 0.043238147161901, "clip_ratio/high_mean": 0.053157048765569925, "clip_ratio/high_max": 0.053157048765569925, "clip_ratio/region_mean": 0.09639519592747092, "reward_total_mean": 0.7046811580657959, "reward_meter_mean": 0.7046811580657959, "reward_meter_std": 0.4017917513847351, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7046811580657959, "reward_total_composite_std": 0.4017917513847351, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 536.0} {"timestamp_utc": "2026-04-11T20:29:53Z", "mode": "train", "global_step": 537, "epoch": 0.020736793327154772, "loss": -0.0426, "grad_norm": 8.848946571350098, "learning_rate": 8.375757575757576e-06, "num_tokens": 1152266.0, "completions/mean_length": 49.75, "completions/min_length": 31.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9934061765670776, "rewards/meter/std": 0.0037812571972608566, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9934061765670776, "rewards/total_composite/std": 0.0037812571972608566, "reward": 0.9934061765670776, "reward_std": 0.003781270468607545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06811977922916412, "sampling/sampling_logp_difference/max": 1.5877177715301514, "sampling/importance_sampling_ratio/min": 0.20439153909683228, "sampling/importance_sampling_ratio/mean": 1.0079725980758667, "sampling/importance_sampling_ratio/max": 1.9671330451965332, "entropy": 0.6782696768641472, "clip_ratio/low_mean": 0.03469064529053867, "clip_ratio/low_min": 0.03469064529053867, "clip_ratio/high_mean": 0.019980506971478462, "clip_ratio/high_max": 0.019980506971478462, "clip_ratio/region_mean": 0.05467115226201713, "reward_total_mean": 0.9934061765670776, "reward_meter_mean": 0.9934061765670776, "reward_meter_std": 0.0037812571972608566, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9934061765670776, "reward_total_composite_std": 0.0037812571972608566, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 537.0} {"timestamp_utc": "2026-04-11T20:29:57Z", "mode": "train", "global_step": 538, "epoch": 0.020775409329626196, "loss": 0.0562, "grad_norm": 11.105710983276367, "learning_rate": 8.372727272727273e-06, "num_tokens": 1153991.0, "completions/mean_length": 60.625, "completions/min_length": 47.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9464359283447266, "rewards/meter/std": 0.10982144623994827, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9464359283447266, "rewards/total_composite/std": 0.10982144623994827, "reward": 0.9464359283447266, "reward_std": 0.10982143878936768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08471371978521347, "sampling/sampling_logp_difference/max": 15.36320972442627, "sampling/importance_sampling_ratio/min": 2.127368787796513e-07, "sampling/importance_sampling_ratio/mean": 1.0101873874664307, "sampling/importance_sampling_ratio/max": 1.6170642375946045, "entropy": 0.5225202403962612, "clip_ratio/low_mean": 0.016544118523597717, "clip_ratio/low_min": 0.016544118523597717, "clip_ratio/high_mean": 0.026057497365400195, "clip_ratio/high_max": 0.026057497365400195, "clip_ratio/region_mean": 0.04260161588899791, "reward_total_mean": 0.9464359283447266, "reward_meter_mean": 0.9464359283447266, "reward_meter_std": 0.10982144623994827, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9464359283447266, "reward_total_composite_std": 0.10982144623994827, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 538.0} {"timestamp_utc": "2026-04-11T20:30:03Z", "mode": "train", "global_step": 539, "epoch": 0.02081402533209762, "loss": 0.0146, "grad_norm": 3.6939918994903564, "learning_rate": 8.36969696969697e-06, "num_tokens": 1155878.0, "completions/mean_length": 63.875, "completions/min_length": 43.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9885964393615723, "rewards/meter/std": 0.016855819150805473, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9885964393615723, "rewards/total_composite/std": 0.016855819150805473, "reward": 0.9885964393615723, "reward_std": 0.01685582473874092, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04388084262609482, "sampling/sampling_logp_difference/max": 0.9602079391479492, "sampling/importance_sampling_ratio/min": 0.3828132748603821, "sampling/importance_sampling_ratio/mean": 1.0144485235214233, "sampling/importance_sampling_ratio/max": 1.6128665208816528, "entropy": 0.5723963566124439, "clip_ratio/low_mean": 0.005681818351149559, "clip_ratio/low_min": 0.005681818351149559, "clip_ratio/high_mean": 0.05073056067340076, "clip_ratio/high_max": 0.05073056067340076, "clip_ratio/region_mean": 0.05641237902455032, "reward_total_mean": 0.9885964393615723, "reward_meter_mean": 0.9885964393615723, "reward_meter_std": 0.016855819150805473, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9885964393615723, "reward_total_composite_std": 0.016855819150805473, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 539.0} {"timestamp_utc": "2026-04-11T20:30:08Z", "mode": "train", "global_step": 540, "epoch": 0.020852641334569044, "loss": 0.1671, "grad_norm": 12.970458030700684, "learning_rate": 8.366666666666667e-06, "num_tokens": 1157571.0, "completions/mean_length": 63.625, "completions/min_length": 55.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.9840399622917175, "rewards/meter/std": 0.00786462053656578, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9218109250068665, "rewards/total_composite/std": 0.1714293360710144, "reward": 0.9218109250068665, "reward_std": 0.1714293360710144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07498990744352341, "sampling/sampling_logp_difference/max": 0.9760599136352539, "sampling/importance_sampling_ratio/min": 0.3767927885055542, "sampling/importance_sampling_ratio/mean": 1.0064505338668823, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48294258303940296, "clip_ratio/low_mean": 0.008426966145634651, "clip_ratio/low_min": 0.008426966145634651, "clip_ratio/high_mean": 0.07365514896810055, "clip_ratio/high_max": 0.07365514896810055, "clip_ratio/region_mean": 0.0820821151137352, "reward_total_mean": 0.9218109250068665, "reward_meter_mean": 0.9840399622917175, "reward_meter_std": 0.00786462053656578, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9218109250068665, "reward_total_composite_std": 0.1714293360710144, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 540.0} {"timestamp_utc": "2026-04-11T20:30:13Z", "mode": "train", "global_step": 541, "epoch": 0.02089125733704047, "loss": -0.0019, "grad_norm": 5.643881797790527, "learning_rate": 8.363636363636365e-06, "num_tokens": 1159414.0, "completions/mean_length": 62.375, "completions/min_length": 55.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9919471740722656, "rewards/meter/std": 0.005765520967543125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9919471740722656, "rewards/total_composite/std": 0.005765520967543125, "reward": 0.9919471740722656, "reward_std": 0.005765520967543125, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.046641286462545395, "sampling/sampling_logp_difference/max": 1.3093020915985107, "sampling/importance_sampling_ratio/min": 0.2700084447860718, "sampling/importance_sampling_ratio/mean": 1.0133535861968994, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2864169627428055, "clip_ratio/low_mean": 0.02023959648795426, "clip_ratio/low_min": 0.02023959648795426, "clip_ratio/high_mean": 0.020859031821601093, "clip_ratio/high_max": 0.020859031821601093, "clip_ratio/region_mean": 0.04109862830955535, "reward_total_mean": 0.9919471740722656, "reward_meter_mean": 0.9919471740722656, "reward_meter_std": 0.005765520967543125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9919471740722656, "reward_total_composite_std": 0.005765520967543125, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 541.0} {"timestamp_utc": "2026-04-11T20:30:19Z", "mode": "train", "global_step": 542, "epoch": 0.020929873339511892, "loss": 0.0263, "grad_norm": 2.8404381275177, "learning_rate": 8.360606060606062e-06, "num_tokens": 1162016.0, "completions/mean_length": 153.25, "completions/min_length": 134.0, "completions/max_length": 177.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 153.25, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 177.0, "rewards/meter/mean": 0.9956228137016296, "rewards/meter/std": 0.0026213540695607662, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9956228137016296, "rewards/total_composite/std": 0.0026213540695607662, "reward": 0.9956228137016296, "reward_std": 0.002621352905407548, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026027433574199677, "sampling/sampling_logp_difference/max": 1.2649219036102295, "sampling/importance_sampling_ratio/min": 0.28226134181022644, "sampling/importance_sampling_ratio/mean": 1.0009527206420898, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2024863502010703, "clip_ratio/low_mean": 0.004344935878179967, "clip_ratio/low_min": 0.004344935878179967, "clip_ratio/high_mean": 0.01289307966362685, "clip_ratio/high_max": 0.01289307966362685, "clip_ratio/region_mean": 0.017238015541806817, "reward_total_mean": 0.9956228137016296, "reward_meter_mean": 0.9956228137016296, "reward_meter_std": 0.0026213540695607662, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9956228137016296, "reward_total_composite_std": 0.0026213540695607662, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 542.0} {"timestamp_utc": "2026-04-11T20:30:28Z", "mode": "train", "global_step": 543, "epoch": 0.020968489341983317, "loss": 0.0722, "grad_norm": 3.049520254135132, "learning_rate": 8.357575757575759e-06, "num_tokens": 1166687.0, "completions/mean_length": 390.875, "completions/min_length": 335.0, "completions/max_length": 463.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 390.875, "completions/min_terminated_length": 335.0, "completions/max_terminated_length": 463.0, "rewards/meter/mean": 0.867215096950531, "rewards/meter/std": 0.3118921220302582, "rewards/count_adherence/mean": 0.453125, "rewards/count_adherence/std": 0.16280877590179443, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3688773512840271, "rewards/total_composite/std": 0.16893041133880615, "reward": 0.3688773512840271, "reward_std": 0.16893039643764496, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011823274195194244, "sampling/sampling_logp_difference/max": 1.2009385824203491, "sampling/importance_sampling_ratio/min": 0.30091166496276855, "sampling/importance_sampling_ratio/mean": 1.0000622272491455, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0655752515885979, "clip_ratio/low_mean": 0.0014523655408993363, "clip_ratio/low_min": 0.0014523655408993363, "clip_ratio/high_mean": 0.008265212236437947, "clip_ratio/high_max": 0.008265212236437947, "clip_ratio/region_mean": 0.009717577777337283, "reward_total_mean": 0.3688773512840271, "reward_meter_mean": 0.867215096950531, "reward_meter_std": 0.3118921220302582, "reward_count_adherence_mean": 0.453125, "reward_count_adherence_std": 0.16280877590179443, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3688773512840271, "reward_total_composite_std": 0.16893041133880615, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 543.0} {"timestamp_utc": "2026-04-11T20:30:34Z", "mode": "train", "global_step": 544, "epoch": 0.02100710534445474, "loss": -0.0199, "grad_norm": 5.046764850616455, "learning_rate": 8.354545454545455e-06, "num_tokens": 1168769.0, "completions/mean_length": 95.25, "completions/min_length": 80.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.25, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9941377639770508, "rewards/meter/std": 0.004864970687776804, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9941377639770508, "rewards/total_composite/std": 0.004864970687776804, "reward": 0.9941377639770508, "reward_std": 0.00486496277153492, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05243554711341858, "sampling/sampling_logp_difference/max": 2.225170612335205, "sampling/importance_sampling_ratio/min": 0.10804898291826248, "sampling/importance_sampling_ratio/mean": 1.0006630420684814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2810734435915947, "clip_ratio/low_mean": 0.035438207909464836, "clip_ratio/low_min": 0.035438207909464836, "clip_ratio/high_mean": 0.019809147692285478, "clip_ratio/high_max": 0.019809147692285478, "clip_ratio/region_mean": 0.055247355601750314, "reward_total_mean": 0.9941377639770508, "reward_meter_mean": 0.9941377639770508, "reward_meter_std": 0.004864970687776804, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9941377639770508, "reward_total_composite_std": 0.004864970687776804, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 544.0} {"timestamp_utc": "2026-04-11T20:30:39Z", "mode": "train", "global_step": 545, "epoch": 0.021045721346926165, "loss": 0.0556, "grad_norm": 5.207396030426025, "learning_rate": 8.351515151515152e-06, "num_tokens": 1170871.0, "completions/mean_length": 98.75, "completions/min_length": 86.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.75, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9043634533882141, "rewards/meter/std": 0.18687696754932404, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9043634533882141, "rewards/total_composite/std": 0.18687696754932404, "reward": 0.9043634533882141, "reward_std": 0.18687698245048523, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03357382118701935, "sampling/sampling_logp_difference/max": 1.1505489349365234, "sampling/importance_sampling_ratio/min": 0.31646302342414856, "sampling/importance_sampling_ratio/mean": 1.0059583187103271, "sampling/importance_sampling_ratio/max": 1.8148711919784546, "entropy": 0.23292037099599838, "clip_ratio/low_mean": 0.004854368977248669, "clip_ratio/low_min": 0.004854368977248669, "clip_ratio/high_mean": 0.018268492887727916, "clip_ratio/high_max": 0.018268492887727916, "clip_ratio/region_mean": 0.023122861864976585, "reward_total_mean": 0.9043634533882141, "reward_meter_mean": 0.9043634533882141, "reward_meter_std": 0.18687696754932404, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9043634533882141, "reward_total_composite_std": 0.18687696754932404, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 545.0} {"timestamp_utc": "2026-04-11T20:30:45Z", "mode": "train", "global_step": 546, "epoch": 0.02108433734939759, "loss": 0.0381, "grad_norm": 4.271132469177246, "learning_rate": 8.348484848484849e-06, "num_tokens": 1173692.0, "completions/mean_length": 168.625, "completions/min_length": 155.0, "completions/max_length": 196.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.625, "completions/min_terminated_length": 155.0, "completions/max_terminated_length": 196.0, "rewards/meter/mean": 0.9413485527038574, "rewards/meter/std": 0.09055134654045105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9413485527038574, "rewards/total_composite/std": 0.09055134654045105, "reward": 0.9413485527038574, "reward_std": 0.09055135399103165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040303874760866165, "sampling/sampling_logp_difference/max": 1.1461381912231445, "sampling/importance_sampling_ratio/min": 0.3178619146347046, "sampling/importance_sampling_ratio/mean": 1.0049318075180054, "sampling/importance_sampling_ratio/max": 1.946368932723999, "entropy": 0.3597529251128435, "clip_ratio/low_mean": 0.015035076532512903, "clip_ratio/low_min": 0.015035076532512903, "clip_ratio/high_mean": 0.011283236730378121, "clip_ratio/high_max": 0.011283236730378121, "clip_ratio/region_mean": 0.026318313262891024, "reward_total_mean": 0.9413485527038574, "reward_meter_mean": 0.9413485527038574, "reward_meter_std": 0.09055134654045105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9413485527038574, "reward_total_composite_std": 0.09055134654045105, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 546.0} {"timestamp_utc": "2026-04-11T20:30:51Z", "mode": "train", "global_step": 547, "epoch": 0.021122953351869013, "loss": -0.0148, "grad_norm": 3.342482566833496, "learning_rate": 8.345454545454546e-06, "num_tokens": 1175997.0, "completions/mean_length": 118.125, "completions/min_length": 111.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.125, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.967841625213623, "rewards/meter/std": 0.06629345566034317, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.967841625213623, "rewards/total_composite/std": 0.06629345566034317, "reward": 0.967841625213623, "reward_std": 0.06629344820976257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0339958630502224, "sampling/sampling_logp_difference/max": 1.9573390483856201, "sampling/importance_sampling_ratio/min": 0.14123374223709106, "sampling/importance_sampling_ratio/mean": 1.0027629137039185, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1798835713416338, "clip_ratio/low_mean": 0.0022522523067891598, "clip_ratio/low_min": 0.0022522523067891598, "clip_ratio/high_mean": 0.032155484426766634, "clip_ratio/high_max": 0.032155484426766634, "clip_ratio/region_mean": 0.034407736733555794, "reward_total_mean": 0.967841625213623, "reward_meter_mean": 0.967841625213623, "reward_meter_std": 0.06629345566034317, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.967841625213623, "reward_total_composite_std": 0.06629345566034317, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 547.0} {"timestamp_utc": "2026-04-11T20:30:57Z", "mode": "train", "global_step": 548, "epoch": 0.021161569354340437, "loss": -0.0406, "grad_norm": 6.153160572052002, "learning_rate": 8.342424242424244e-06, "num_tokens": 1178566.0, "completions/mean_length": 133.125, "completions/min_length": 102.0, "completions/max_length": 179.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.125, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 179.0, "rewards/meter/mean": 0.9785022139549255, "rewards/meter/std": 0.02502669021487236, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9537743330001831, "rewards/total_composite/std": 0.07013029605150223, "reward": 0.9537743330001831, "reward_std": 0.07013029605150223, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035310305655002594, "sampling/sampling_logp_difference/max": 1.1522560119628906, "sampling/importance_sampling_ratio/min": 0.3159232437610626, "sampling/importance_sampling_ratio/mean": 1.0059552192687988, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2751317787915468, "clip_ratio/low_mean": 0.011371727799996734, "clip_ratio/low_min": 0.011371727799996734, "clip_ratio/high_mean": 0.025215634261257946, "clip_ratio/high_max": 0.025215634261257946, "clip_ratio/region_mean": 0.03658736206125468, "reward_total_mean": 0.9537743330001831, "reward_meter_mean": 0.9785022139549255, "reward_meter_std": 0.02502669021487236, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9537743330001831, "reward_total_composite_std": 0.07013029605150223, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 548.0} {"timestamp_utc": "2026-04-11T20:31:01Z", "mode": "train", "global_step": 549, "epoch": 0.02120018535681186, "loss": -0.077, "grad_norm": 17.468664169311523, "learning_rate": 8.339393939393941e-06, "num_tokens": 1180024.0, "completions/mean_length": 29.25, "completions/min_length": 19.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.25, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.989723801612854, "rewards/meter/std": 0.011034172028303146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.989723801612854, "rewards/total_composite/std": 0.011034172028303146, "reward": 0.989723801612854, "reward_std": 0.011034182272851467, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10392901301383972, "sampling/sampling_logp_difference/max": 1.3459672927856445, "sampling/importance_sampling_ratio/min": 0.2602878212928772, "sampling/importance_sampling_ratio/mean": 1.0076111555099487, "sampling/importance_sampling_ratio/max": 1.7388663291931152, "entropy": 0.9557730779051781, "clip_ratio/low_mean": 0.05864309147000313, "clip_ratio/low_min": 0.05864309147000313, "clip_ratio/high_mean": 0.04026081250049174, "clip_ratio/high_max": 0.04026081250049174, "clip_ratio/region_mean": 0.09890390397049487, "reward_total_mean": 0.989723801612854, "reward_meter_mean": 0.989723801612854, "reward_meter_std": 0.011034172028303146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.989723801612854, "reward_total_composite_std": 0.011034172028303146, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 549.0} {"timestamp_utc": "2026-04-11T20:31:12Z", "mode": "train", "global_step": 550, "epoch": 0.021238801359283285, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 8.336363636363636e-06, "num_tokens": 1181720.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9942543506622314, "rewards/meter/std": 0.005934838205575943, "rewards/count_adherence/mean": 0.9191176891326904, "rewards/count_adherence/std": 0.05388972535729408, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9139711856842041, "rewards/total_composite/std": 0.05620856583118439, "reward": 0.9139711856842041, "reward_std": 0.056208569556474686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9139711856842041, "reward_meter_mean": 0.9942543506622314, "reward_meter_std": 0.005934838205575943, "reward_count_adherence_mean": 0.9191176891326904, "reward_count_adherence_std": 0.05388972535729408, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9139711856842041, "reward_total_composite_std": 0.05620856583118439, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 550.0} {"timestamp_utc": "2026-04-11T20:32:40Z", "mode": "eval", "global_step": 550, "epoch": 0.021238801359283285, "eval_loss": NaN, "eval_runtime": 88.1689, "eval_samples_per_second": 1.18, "eval_steps_per_second": 0.147, "eval_num_tokens": 1181720.0, "eval_completions/mean_length": 241.0096153846154, "eval_completions/min_length": 51.69230769230769, "eval_completions/max_length": 476.38461538461536, "eval_completions/clipped_ratio": 0.125, "eval_completions/mean_terminated_length": 199.7985393817608, "eval_completions/min_terminated_length": 51.69230769230769, "eval_completions/max_terminated_length": 421.0769230769231, "eval_rewards/meter/mean": 0.7935248292409457, "eval_rewards/meter/std": 0.3153245966308392, "eval_rewards/count_adherence/mean": 0.8751364029370822, "eval_rewards/count_adherence/std": 0.1485061846100367, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/total_composite/mean": 0.7004609107971191, "eval_rewards/total_composite/std": 0.3202407302764746, "eval_reward": 0.7004609107971191, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.014547023384903487, "eval_sampling/sampling_logp_difference/max": 0.8992016132061298, "eval_sampling/importance_sampling_ratio/min": 0.41424351701369655, "eval_sampling/importance_sampling_ratio/mean": 1.0041714814993052, "eval_sampling/importance_sampling_ratio/max": 1.3832710614571204, "eval_entropy": 0.1650333639520865, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7004609107971191, "eval_reward_meter_mean": 0.7935248292409457, "eval_reward_meter_std": 0.3153245966308392, "eval_reward_count_adherence_mean": 0.8751364029370822, "eval_reward_count_adherence_std": 0.1485061846100367, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_total_composite_mean": 0.7004609107971191, "eval_reward_total_composite_std": 0.3202407302764746, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 550.0} {"timestamp_utc": "2026-04-11T20:32:52Z", "mode": "train", "global_step": 551, "epoch": 0.021277417361754713, "loss": -0.1884, "grad_norm": 1.8348972797393799, "learning_rate": 8.333333333333334e-06, "num_tokens": 1184507.0, "completions/mean_length": 234.375, "completions/min_length": 169.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 194.71429443359375, "completions/min_terminated_length": 169.0, "completions/max_terminated_length": 217.0, "rewards/meter/mean": 0.659835934638977, "rewards/meter/std": 0.40453797578811646, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.637830376625061, "rewards/total_composite/std": 0.39101067185401917, "reward": 0.637830376625061, "reward_std": 0.39101067185401917, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03236944600939751, "sampling/sampling_logp_difference/max": 1.1597709655761719, "sampling/importance_sampling_ratio/min": 0.31355801224708557, "sampling/importance_sampling_ratio/mean": 1.002615213394165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2244832031428814, "clip_ratio/low_mean": 0.004437869880348444, "clip_ratio/low_min": 0.004437869880348444, "clip_ratio/high_mean": 0.02020994306076318, "clip_ratio/high_max": 0.02020994306076318, "clip_ratio/region_mean": 0.024647812941111624, "reward_total_mean": 0.637830376625061, "reward_meter_mean": 0.659835934638977, "reward_meter_std": 0.40453797578811646, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.637830376625061, "reward_total_composite_std": 0.39101067185401917, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 551.0} {"timestamp_utc": "2026-04-11T20:32:58Z", "mode": "train", "global_step": 552, "epoch": 0.021316033364226137, "loss": 0.015, "grad_norm": 2.554471731185913, "learning_rate": 8.330303030303031e-06, "num_tokens": 1187017.0, "completions/mean_length": 147.75, "completions/min_length": 129.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.75, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.9956350326538086, "rewards/meter/std": 0.0016549181891605258, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7965080738067627, "rewards/total_composite/std": 0.0013239418622106314, "reward": 0.7965080738067627, "reward_std": 0.0013239351101219654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.037060461938381195, "sampling/sampling_logp_difference/max": 1.7529680728912354, "sampling/importance_sampling_ratio/min": 0.1732589304447174, "sampling/importance_sampling_ratio/mean": 1.0011682510375977, "sampling/importance_sampling_ratio/max": 1.891858458518982, "entropy": 0.31488965079188347, "clip_ratio/low_mean": 0.012516135815531015, "clip_ratio/low_min": 0.012516135815531015, "clip_ratio/high_mean": 0.027031495235860348, "clip_ratio/high_max": 0.027031495235860348, "clip_ratio/region_mean": 0.03954763105139136, "reward_total_mean": 0.7965080738067627, "reward_meter_mean": 0.9956350326538086, "reward_meter_std": 0.0016549181891605258, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7965080738067627, "reward_total_composite_std": 0.0013239418622106314, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 552.0} {"timestamp_utc": "2026-04-11T20:33:09Z", "mode": "train", "global_step": 553, "epoch": 0.02135464936669756, "loss": 0.1244, "grad_norm": 0.24584703147411346, "learning_rate": 8.327272727272728e-06, "num_tokens": 1190519.0, "completions/mean_length": 509.75, "completions/min_length": 497.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 506.0, "completions/min_terminated_length": 497.0, "completions/max_terminated_length": 511.0, "rewards/meter/mean": 0.9953852891921997, "rewards/meter/std": 0.00345088099129498, "rewards/count_adherence/mean": 0.7708333730697632, "rewards/count_adherence/std": 0.058925554156303406, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.767181932926178, "rewards/total_composite/std": 0.05727330222725868, "reward": 0.767181932926178, "reward_std": 0.05727332457900047, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008550068363547325, "sampling/sampling_logp_difference/max": 1.183694839477539, "sampling/importance_sampling_ratio/min": 0.30614548921585083, "sampling/importance_sampling_ratio/mean": 1.0002206563949585, "sampling/importance_sampling_ratio/max": 1.871092677116394, "entropy": 0.021786156576126814, "clip_ratio/low_mean": 0.003231151611544192, "clip_ratio/low_min": 0.003231151611544192, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.003231151611544192, "reward_total_mean": 0.767181932926178, "reward_meter_mean": 0.9953852891921997, "reward_meter_std": 0.00345088099129498, "reward_count_adherence_mean": 0.7708333730697632, "reward_count_adherence_std": 0.058925554156303406, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.767181932926178, "reward_total_composite_std": 0.05727330222725868, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 553.0} {"timestamp_utc": "2026-04-11T20:33:19Z", "mode": "train", "global_step": 554, "epoch": 0.021393265369168985, "loss": 0.378, "grad_norm": 1.8809558153152466, "learning_rate": 8.324242424242425e-06, "num_tokens": 1193244.0, "completions/mean_length": 504.625, "completions/min_length": 468.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.75, "completions/mean_terminated_length": 482.5, "completions/min_terminated_length": 468.0, "completions/max_terminated_length": 497.0, "rewards/meter/mean": 0.7643306255340576, "rewards/meter/std": 0.37500372529029846, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1075671836733818, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6255342364311218, "rewards/total_composite/std": 0.32354018092155457, "reward": 0.6255342364311218, "reward_std": 0.32354021072387695, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06437436491250992, "sampling/sampling_logp_difference/max": 1.3047142028808594, "sampling/importance_sampling_ratio/min": 0.27125003933906555, "sampling/importance_sampling_ratio/mean": 1.0125058889389038, "sampling/importance_sampling_ratio/max": 1.7761412858963013, "entropy": 0.19928418099880219, "clip_ratio/low_mean": 0.0089119115145877, "clip_ratio/low_min": 0.0089119115145877, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0089119115145877, "reward_total_mean": 0.6255342364311218, "reward_meter_mean": 0.7643306255340576, "reward_meter_std": 0.37500372529029846, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1075671836733818, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6255342364311218, "reward_total_composite_std": 0.32354018092155457, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 554.0} {"timestamp_utc": "2026-04-11T20:33:25Z", "mode": "train", "global_step": 555, "epoch": 0.02143188137164041, "loss": -0.0045, "grad_norm": 2.518230676651001, "learning_rate": 8.321212121212123e-06, "num_tokens": 1195591.0, "completions/mean_length": 126.375, "completions/min_length": 121.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.375, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.9948133826255798, "rewards/meter/std": 0.0038075658958405256, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.839267909526825, "rewards/total_composite/std": 0.12791672348976135, "reward": 0.839267909526825, "reward_std": 0.12791673839092255, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0339997336268425, "sampling/sampling_logp_difference/max": 1.2885704040527344, "sampling/importance_sampling_ratio/min": 0.2756645977497101, "sampling/importance_sampling_ratio/mean": 1.0079847574234009, "sampling/importance_sampling_ratio/max": 1.7961597442626953, "entropy": 0.2988105323165655, "clip_ratio/low_mean": 0.017126905964687467, "clip_ratio/low_min": 0.017126905964687467, "clip_ratio/high_mean": 0.005685312673449516, "clip_ratio/high_max": 0.005685312673449516, "clip_ratio/region_mean": 0.022812218638136983, "reward_total_mean": 0.839267909526825, "reward_meter_mean": 0.9948133826255798, "reward_meter_std": 0.0038075658958405256, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.839267909526825, "reward_total_composite_std": 0.12791672348976135, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 555.0} {"timestamp_utc": "2026-04-11T20:33:35Z", "mode": "train", "global_step": 556, "epoch": 0.021470497374111833, "loss": 0.0222, "grad_norm": 1.6413999795913696, "learning_rate": 8.318181818181818e-06, "num_tokens": 1200948.0, "completions/mean_length": 482.625, "completions/min_length": 461.0, "completions/max_length": 500.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 482.625, "completions/min_terminated_length": 461.0, "completions/max_terminated_length": 500.0, "rewards/meter/mean": 0.9967899918556213, "rewards/meter/std": 0.0011801483342424035, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.0589255690574646, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8098647594451904, "rewards/total_composite/std": 0.058309562504291534, "reward": 0.8098647594451904, "reward_std": 0.058309562504291534, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010928530246019363, "sampling/sampling_logp_difference/max": 1.6081576347351074, "sampling/importance_sampling_ratio/min": 0.20025621354579926, "sampling/importance_sampling_ratio/mean": 1.0008271932601929, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06857710564509034, "clip_ratio/low_mean": 0.0012622967187780887, "clip_ratio/low_min": 0.0012622967187780887, "clip_ratio/high_mean": 0.007534351840149611, "clip_ratio/high_max": 0.007534351840149611, "clip_ratio/region_mean": 0.0087966485589277, "reward_total_mean": 0.8098647594451904, "reward_meter_mean": 0.9967899918556213, "reward_meter_std": 0.0011801483342424035, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.0589255690574646, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8098647594451904, "reward_total_composite_std": 0.058309562504291534, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 556.0} {"timestamp_utc": "2026-04-11T20:33:40Z", "mode": "train", "global_step": 557, "epoch": 0.021509113376583257, "loss": 0.0447, "grad_norm": 12.06613826751709, "learning_rate": 8.315151515151516e-06, "num_tokens": 1203421.0, "completions/mean_length": 109.125, "completions/min_length": 95.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.125, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.5895034074783325, "rewards/meter/std": 0.44556495547294617, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5651826858520508, "rewards/total_composite/std": 0.42655694484710693, "reward": 0.5651826858520508, "reward_std": 0.42655691504478455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.057171959429979324, "sampling/sampling_logp_difference/max": 2.000739097595215, "sampling/importance_sampling_ratio/min": 0.13523530960083008, "sampling/importance_sampling_ratio/mean": 0.9983038306236267, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2295174626633525, "clip_ratio/low_mean": 0.026261302642524242, "clip_ratio/low_min": 0.026261302642524242, "clip_ratio/high_mean": 0.024866676423698664, "clip_ratio/high_max": 0.024866676423698664, "clip_ratio/region_mean": 0.051127979066222906, "reward_total_mean": 0.5651826858520508, "reward_meter_mean": 0.5895034074783325, "reward_meter_std": 0.44556495547294617, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5651826858520508, "reward_total_composite_std": 0.42655694484710693, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 557.0} {"timestamp_utc": "2026-04-11T20:33:45Z", "mode": "train", "global_step": 558, "epoch": 0.02154772937905468, "loss": -0.019, "grad_norm": 4.484973907470703, "learning_rate": 8.312121212121213e-06, "num_tokens": 1205157.0, "completions/mean_length": 61.0, "completions/min_length": 58.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9771539568901062, "rewards/meter/std": 0.036376286298036575, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9771539568901062, "rewards/total_composite/std": 0.036376286298036575, "reward": 0.9771539568901062, "reward_std": 0.03637627884745598, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04534757509827614, "sampling/sampling_logp_difference/max": 1.7598295211791992, "sampling/importance_sampling_ratio/min": 0.17207419872283936, "sampling/importance_sampling_ratio/mean": 0.9910410046577454, "sampling/importance_sampling_ratio/max": 1.4410507678985596, "entropy": 0.24670910649001598, "clip_ratio/low_mean": 0.010358731728047132, "clip_ratio/low_min": 0.010358731728047132, "clip_ratio/high_mean": 0.025330669013783336, "clip_ratio/high_max": 0.025330669013783336, "clip_ratio/region_mean": 0.03568940074183047, "reward_total_mean": 0.9771539568901062, "reward_meter_mean": 0.9771539568901062, "reward_meter_std": 0.036376286298036575, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9771539568901062, "reward_total_composite_std": 0.036376286298036575, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 558.0} {"timestamp_utc": "2026-04-11T20:33:50Z", "mode": "train", "global_step": 559, "epoch": 0.021586345381526106, "loss": 0.0143, "grad_norm": 12.47499942779541, "learning_rate": 8.30909090909091e-06, "num_tokens": 1206934.0, "completions/mean_length": 55.125, "completions/min_length": 52.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9845154285430908, "rewards/meter/std": 0.022365685552358627, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9845154285430908, "rewards/total_composite/std": 0.022365685552358627, "reward": 0.9845154285430908, "reward_std": 0.022365683689713478, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05525405332446098, "sampling/sampling_logp_difference/max": 1.1816325187683105, "sampling/importance_sampling_ratio/min": 0.3067775070667267, "sampling/importance_sampling_ratio/mean": 1.0180634260177612, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3850720599293709, "clip_ratio/low_mean": 0.013174113817512989, "clip_ratio/low_min": 0.013174113817512989, "clip_ratio/high_mean": 0.027243590680882335, "clip_ratio/high_max": 0.027243590680882335, "clip_ratio/region_mean": 0.040417704498395324, "reward_total_mean": 0.9845154285430908, "reward_meter_mean": 0.9845154285430908, "reward_meter_std": 0.022365685552358627, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9845154285430908, "reward_total_composite_std": 0.022365685552358627, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 559.0} {"timestamp_utc": "2026-04-11T20:33:55Z", "mode": "train", "global_step": 560, "epoch": 0.02162496138399753, "loss": -0.0201, "grad_norm": 7.248865604400635, "learning_rate": 8.306060606060606e-06, "num_tokens": 1209005.0, "completions/mean_length": 86.875, "completions/min_length": 79.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.875, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.923915684223175, "rewards/meter/std": 0.18663685023784637, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.923915684223175, "rewards/total_composite/std": 0.18663685023784637, "reward": 0.923915684223175, "reward_std": 0.18663683533668518, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03298182412981987, "sampling/sampling_logp_difference/max": 1.1690406799316406, "sampling/importance_sampling_ratio/min": 0.3106648325920105, "sampling/importance_sampling_ratio/mean": 1.0030776262283325, "sampling/importance_sampling_ratio/max": 1.8499445915222168, "entropy": 0.22090137097984552, "clip_ratio/low_mean": 0.004687500186264515, "clip_ratio/low_min": 0.004687500186264515, "clip_ratio/high_mean": 0.029417280456982553, "clip_ratio/high_max": 0.029417280456982553, "clip_ratio/region_mean": 0.03410478064324707, "reward_total_mean": 0.923915684223175, "reward_meter_mean": 0.923915684223175, "reward_meter_std": 0.18663685023784637, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.923915684223175, "reward_total_composite_std": 0.18663685023784637, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 560.0} {"timestamp_utc": "2026-04-11T20:34:00Z", "mode": "train", "global_step": 561, "epoch": 0.021663577386468954, "loss": -0.009, "grad_norm": 5.603593826293945, "learning_rate": 8.303030303030305e-06, "num_tokens": 1211172.0, "completions/mean_length": 96.875, "completions/min_length": 91.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.875, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.995019793510437, "rewards/meter/std": 0.0025853195693343878, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9536159038543701, "rewards/total_composite/std": 0.11767086386680603, "reward": 0.9536159038543701, "reward_std": 0.11767084896564484, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05422169715166092, "sampling/sampling_logp_difference/max": 2.6318905353546143, "sampling/importance_sampling_ratio/min": 0.07194232940673828, "sampling/importance_sampling_ratio/mean": 1.0120702981948853, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5123469866812229, "clip_ratio/low_mean": 0.005494505632668734, "clip_ratio/low_min": 0.005494505632668734, "clip_ratio/high_mean": 0.04831800376996398, "clip_ratio/high_max": 0.04831800376996398, "clip_ratio/region_mean": 0.05381250940263271, "reward_total_mean": 0.9536159038543701, "reward_meter_mean": 0.995019793510437, "reward_meter_std": 0.0025853195693343878, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9536159038543701, "reward_total_composite_std": 0.11767086386680603, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 561.0} {"timestamp_utc": "2026-04-11T20:34:05Z", "mode": "train", "global_step": 562, "epoch": 0.021702193388940378, "loss": 0.2898, "grad_norm": 11.571051597595215, "learning_rate": 8.3e-06, "num_tokens": 1212666.0, "completions/mean_length": 37.75, "completions/min_length": 26.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9433770179748535, "rewards/meter/std": 0.13426972925662994, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8193449378013611, "rewards/total_composite/std": 0.3567107319831848, "reward": 0.8193449378013611, "reward_std": 0.3567107319831848, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09738308191299438, "sampling/sampling_logp_difference/max": 1.4824190139770508, "sampling/importance_sampling_ratio/min": 0.22708770632743835, "sampling/importance_sampling_ratio/mean": 1.0174795389175415, "sampling/importance_sampling_ratio/max": 1.9018025398254395, "entropy": 0.7698529586195946, "clip_ratio/low_mean": 0.0078125, "clip_ratio/low_min": 0.0078125, "clip_ratio/high_mean": 0.06236687349155545, "clip_ratio/high_max": 0.06236687349155545, "clip_ratio/region_mean": 0.07017937349155545, "reward_total_mean": 0.8193449378013611, "reward_meter_mean": 0.9433770179748535, "reward_meter_std": 0.13426972925662994, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8193449378013611, "reward_total_composite_std": 0.3567107319831848, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 562.0} {"timestamp_utc": "2026-04-11T20:34:11Z", "mode": "train", "global_step": 563, "epoch": 0.021740809391411802, "loss": 0.0224, "grad_norm": 4.293374061584473, "learning_rate": 8.296969696969697e-06, "num_tokens": 1214907.0, "completions/mean_length": 100.125, "completions/min_length": 96.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.125, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.7916634678840637, "rewards/meter/std": 0.35509946942329407, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7916634678840637, "rewards/total_composite/std": 0.35509946942329407, "reward": 0.7916634678840637, "reward_std": 0.3550994396209717, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04964831843972206, "sampling/sampling_logp_difference/max": 0.95345139503479, "sampling/importance_sampling_ratio/min": 0.38540852069854736, "sampling/importance_sampling_ratio/mean": 1.0102379322052002, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4031477700918913, "clip_ratio/low_mean": 0.008614832302555442, "clip_ratio/low_min": 0.008614832302555442, "clip_ratio/high_mean": 0.02509535406716168, "clip_ratio/high_max": 0.02509535406716168, "clip_ratio/region_mean": 0.03371018636971712, "reward_total_mean": 0.7916634678840637, "reward_meter_mean": 0.7916634678840637, "reward_meter_std": 0.35509946942329407, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7916634678840637, "reward_total_composite_std": 0.35509946942329407, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 563.0} {"timestamp_utc": "2026-04-11T20:34:16Z", "mode": "train", "global_step": 564, "epoch": 0.021779425393883226, "loss": -0.1095, "grad_norm": 7.8068928718566895, "learning_rate": 8.293939393939395e-06, "num_tokens": 1216596.0, "completions/mean_length": 59.125, "completions/min_length": 43.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9822205305099487, "rewards/meter/std": 0.0145874610170722, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9822205305099487, "rewards/total_composite/std": 0.0145874610170722, "reward": 0.9822205305099487, "reward_std": 0.014587470330297947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0998057872056961, "sampling/sampling_logp_difference/max": 2.658904552459717, "sampling/importance_sampling_ratio/min": 0.07002489268779755, "sampling/importance_sampling_ratio/mean": 1.0125072002410889, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.574961706995964, "clip_ratio/low_mean": 0.031700728461146355, "clip_ratio/low_min": 0.031700728461146355, "clip_ratio/high_mean": 0.044469698797911406, "clip_ratio/high_max": 0.044469698797911406, "clip_ratio/region_mean": 0.07617042725905776, "reward_total_mean": 0.9822205305099487, "reward_meter_mean": 0.9822205305099487, "reward_meter_std": 0.0145874610170722, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9822205305099487, "reward_total_composite_std": 0.0145874610170722, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 564.0} {"timestamp_utc": "2026-04-11T20:34:20Z", "mode": "train", "global_step": 565, "epoch": 0.02181804139635465, "loss": 0.1893, "grad_norm": 9.879420280456543, "learning_rate": 8.290909090909092e-06, "num_tokens": 1218119.0, "completions/mean_length": 36.375, "completions/min_length": 29.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7298904061317444, "rewards/meter/std": 0.45080265402793884, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7292733192443848, "rewards/total_composite/std": 0.4519386887550354, "reward": 0.7292733192443848, "reward_std": 0.451938658952713, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09998878836631775, "sampling/sampling_logp_difference/max": 1.0424585342407227, "sampling/importance_sampling_ratio/min": 0.3525867760181427, "sampling/importance_sampling_ratio/mean": 1.0224130153656006, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8591618165373802, "clip_ratio/low_mean": 0.021110056899487972, "clip_ratio/low_min": 0.021110056899487972, "clip_ratio/high_mean": 0.056833125185221434, "clip_ratio/high_max": 0.056833125185221434, "clip_ratio/region_mean": 0.0779431820847094, "reward_total_mean": 0.7292733192443848, "reward_meter_mean": 0.7298904061317444, "reward_meter_std": 0.45080265402793884, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7292733192443848, "reward_total_composite_std": 0.4519386887550354, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 565.0} {"timestamp_utc": "2026-04-11T20:34:25Z", "mode": "train", "global_step": 566, "epoch": 0.021856657398826074, "loss": -0.0217, "grad_norm": 7.685236930847168, "learning_rate": 8.287878787878787e-06, "num_tokens": 1220231.0, "completions/mean_length": 73.0, "completions/min_length": 66.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8192075490951538, "rewards/meter/std": 0.1594366729259491, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8192075490951538, "rewards/total_composite/std": 0.1594366729259491, "reward": 0.8192075490951538, "reward_std": 0.1594366729259491, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07931912690401077, "sampling/sampling_logp_difference/max": 1.4172749519348145, "sampling/importance_sampling_ratio/min": 0.24237361550331116, "sampling/importance_sampling_ratio/mean": 1.019374132156372, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5164803229272366, "clip_ratio/low_mean": 0.03550754114985466, "clip_ratio/low_min": 0.03550754114985466, "clip_ratio/high_mean": 0.019384657382033765, "clip_ratio/high_max": 0.019384657382033765, "clip_ratio/region_mean": 0.054892198531888425, "reward_total_mean": 0.8192075490951538, "reward_meter_mean": 0.8192075490951538, "reward_meter_std": 0.1594366729259491, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8192075490951538, "reward_total_composite_std": 0.1594366729259491, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 566.0} {"timestamp_utc": "2026-04-11T20:34:30Z", "mode": "train", "global_step": 567, "epoch": 0.0218952734012975, "loss": 0.0125, "grad_norm": 13.685770988464355, "learning_rate": 8.284848484848486e-06, "num_tokens": 1221897.0, "completions/mean_length": 48.25, "completions/min_length": 44.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.25, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7687733173370361, "rewards/meter/std": 0.23431365191936493, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7687733173370361, "rewards/total_composite/std": 0.23431365191936493, "reward": 0.7687733173370361, "reward_std": 0.23431365191936493, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12316906452178955, "sampling/sampling_logp_difference/max": 1.8604542016983032, "sampling/importance_sampling_ratio/min": 0.15560193359851837, "sampling/importance_sampling_ratio/mean": 1.0146090984344482, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7732437737286091, "clip_ratio/low_mean": 0.06415719725191593, "clip_ratio/low_min": 0.06415719725191593, "clip_ratio/high_mean": 0.05356209189631045, "clip_ratio/high_max": 0.05356209189631045, "clip_ratio/region_mean": 0.11771928914822638, "reward_total_mean": 0.7687733173370361, "reward_meter_mean": 0.7687733173370361, "reward_meter_std": 0.23431365191936493, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7687733173370361, "reward_total_composite_std": 0.23431365191936493, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 567.0} {"timestamp_utc": "2026-04-11T20:34:35Z", "mode": "train", "global_step": 568, "epoch": 0.021933889403768923, "loss": 0.0232, "grad_norm": 12.740221977233887, "learning_rate": 8.281818181818182e-06, "num_tokens": 1223686.0, "completions/mean_length": 41.625, "completions/min_length": 35.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.3464130163192749, "rewards/meter/std": 0.327831894159317, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3464130163192749, "rewards/total_composite/std": 0.327831894159317, "reward": 0.3464130163192749, "reward_std": 0.327831894159317, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13335032761096954, "sampling/sampling_logp_difference/max": 1.874311089515686, "sampling/importance_sampling_ratio/min": 0.15346065163612366, "sampling/importance_sampling_ratio/mean": 1.0059748888015747, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1647358424961567, "clip_ratio/low_mean": 0.07042828761041164, "clip_ratio/low_min": 0.07042828761041164, "clip_ratio/high_mean": 0.07650120463222265, "clip_ratio/high_max": 0.07650120463222265, "clip_ratio/region_mean": 0.1469294922426343, "reward_total_mean": 0.3464130163192749, "reward_meter_mean": 0.3464130163192749, "reward_meter_std": 0.327831894159317, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3464130163192749, "reward_total_composite_std": 0.327831894159317, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 568.0} {"timestamp_utc": "2026-04-11T20:34:43Z", "mode": "train", "global_step": 569, "epoch": 0.021972505406240347, "loss": 0.0953, "grad_norm": 5.267251014709473, "learning_rate": 8.27878787878788e-06, "num_tokens": 1226918.0, "completions/mean_length": 208.0, "completions/min_length": 151.0, "completions/max_length": 263.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 208.0, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 263.0, "rewards/meter/mean": 0.8887746334075928, "rewards/meter/std": 0.28811898827552795, "rewards/count_adherence/mean": 0.9642857313156128, "rewards/count_adherence/std": 0.06613000482320786, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8531581163406372, "rewards/total_composite/std": 0.28023195266723633, "reward": 0.8531581163406372, "reward_std": 0.28023195266723633, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04299961030483246, "sampling/sampling_logp_difference/max": 2.6310670375823975, "sampling/importance_sampling_ratio/min": 0.07200159877538681, "sampling/importance_sampling_ratio/mean": 1.004589319229126, "sampling/importance_sampling_ratio/max": 1.8939993381500244, "entropy": 0.30806298181414604, "clip_ratio/low_mean": 0.0090304184705019, "clip_ratio/low_min": 0.0090304184705019, "clip_ratio/high_mean": 0.020774202537722886, "clip_ratio/high_max": 0.020774202537722886, "clip_ratio/region_mean": 0.029804621008224785, "reward_total_mean": 0.8531581163406372, "reward_meter_mean": 0.8887746334075928, "reward_meter_std": 0.28811898827552795, "reward_count_adherence_mean": 0.9642857313156128, "reward_count_adherence_std": 0.06613000482320786, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8531581163406372, "reward_total_composite_std": 0.28023195266723633, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 569.0} {"timestamp_utc": "2026-04-11T20:34:48Z", "mode": "train", "global_step": 570, "epoch": 0.02201112140871177, "loss": 0.0113, "grad_norm": 4.654249668121338, "learning_rate": 8.275757575757577e-06, "num_tokens": 1228806.0, "completions/mean_length": 65.0, "completions/min_length": 62.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.984466016292572, "rewards/meter/std": 0.027155917137861252, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.984466016292572, "rewards/total_composite/std": 0.027155917137861252, "reward": 0.984466016292572, "reward_std": 0.027155913412570953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.053486477583646774, "sampling/sampling_logp_difference/max": 1.147278070449829, "sampling/importance_sampling_ratio/min": 0.3174997866153717, "sampling/importance_sampling_ratio/mean": 1.0072888135910034, "sampling/importance_sampling_ratio/max": 1.9392898082733154, "entropy": 0.3545149974524975, "clip_ratio/low_mean": 0.005681818351149559, "clip_ratio/low_min": 0.005681818351149559, "clip_ratio/high_mean": 0.04666911787353456, "clip_ratio/high_max": 0.04666911787353456, "clip_ratio/region_mean": 0.05235093622468412, "reward_total_mean": 0.984466016292572, "reward_meter_mean": 0.984466016292572, "reward_meter_std": 0.027155917137861252, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.984466016292572, "reward_total_composite_std": 0.027155917137861252, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 570.0} {"timestamp_utc": "2026-04-11T20:34:57Z", "mode": "train", "global_step": 571, "epoch": 0.022049737411183195, "loss": -0.01, "grad_norm": 2.5224688053131104, "learning_rate": 8.272727272727274e-06, "num_tokens": 1233705.0, "completions/mean_length": 372.375, "completions/min_length": 329.0, "completions/max_length": 419.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 372.375, "completions/min_terminated_length": 329.0, "completions/max_terminated_length": 419.0, "rewards/meter/mean": 0.5269562005996704, "rewards/meter/std": 0.4027676284313202, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.09234060347080231, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3962605595588684, "rewards/total_composite/std": 0.30246567726135254, "reward": 0.3962605595588684, "reward_std": 0.30246567726135254, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0314236655831337, "sampling/sampling_logp_difference/max": 2.1823997497558594, "sampling/importance_sampling_ratio/min": 0.11277057975530624, "sampling/importance_sampling_ratio/mean": 0.9997126460075378, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16989378002472222, "clip_ratio/low_mean": 0.015158275258727372, "clip_ratio/low_min": 0.015158275258727372, "clip_ratio/high_mean": 0.014448553905822337, "clip_ratio/high_max": 0.014448553905822337, "clip_ratio/region_mean": 0.02960682916454971, "reward_total_mean": 0.3962605595588684, "reward_meter_mean": 0.5269562005996704, "reward_meter_std": 0.4027676284313202, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.09234060347080231, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3962605595588684, "reward_total_composite_std": 0.30246567726135254, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 571.0} {"timestamp_utc": "2026-04-11T20:35:02Z", "mode": "train", "global_step": 572, "epoch": 0.02208835341365462, "loss": -0.0052, "grad_norm": 8.577203750610352, "learning_rate": 8.269696969696971e-06, "num_tokens": 1235635.0, "completions/mean_length": 79.25, "completions/min_length": 74.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.25, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.6216261386871338, "rewards/meter/std": 0.42476242780685425, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6216261386871338, "rewards/total_composite/std": 0.42476242780685425, "reward": 0.6216261386871338, "reward_std": 0.42476242780685425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08498334139585495, "sampling/sampling_logp_difference/max": 1.9287548065185547, "sampling/importance_sampling_ratio/min": 0.27314189076423645, "sampling/importance_sampling_ratio/mean": 1.0138676166534424, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6656168103218079, "clip_ratio/low_mean": 0.025765900732949376, "clip_ratio/low_min": 0.025765900732949376, "clip_ratio/high_mean": 0.04550157627090812, "clip_ratio/high_max": 0.04550157627090812, "clip_ratio/region_mean": 0.0712674770038575, "reward_total_mean": 0.6216261386871338, "reward_meter_mean": 0.6216261386871338, "reward_meter_std": 0.42476242780685425, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6216261386871338, "reward_total_composite_std": 0.42476242780685425, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 572.0} {"timestamp_utc": "2026-04-11T20:35:06Z", "mode": "train", "global_step": 573, "epoch": 0.022126969416126043, "loss": -0.0174, "grad_norm": 14.97447681427002, "learning_rate": 8.266666666666667e-06, "num_tokens": 1237082.0, "completions/mean_length": 28.875, "completions/min_length": 25.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.8920953273773193, "rewards/meter/std": 0.2680381238460541, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8920953273773193, "rewards/total_composite/std": 0.2680381238460541, "reward": 0.8920953273773193, "reward_std": 0.2680380940437317, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12218277901411057, "sampling/sampling_logp_difference/max": 1.061295509338379, "sampling/importance_sampling_ratio/min": 0.3460072875022888, "sampling/importance_sampling_ratio/mean": 1.0461758375167847, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1661205813288689, "clip_ratio/low_mean": 0.023148147389292717, "clip_ratio/low_min": 0.023148147389292717, "clip_ratio/high_mean": 0.08289422187954187, "clip_ratio/high_max": 0.08289422187954187, "clip_ratio/region_mean": 0.10604236926883459, "reward_total_mean": 0.8920953273773193, "reward_meter_mean": 0.8920953273773193, "reward_meter_std": 0.2680381238460541, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8920953273773193, "reward_total_composite_std": 0.2680381238460541, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 573.0} {"timestamp_utc": "2026-04-11T20:35:13Z", "mode": "train", "global_step": 574, "epoch": 0.022165585418597467, "loss": 0.0108, "grad_norm": 1.9361296892166138, "learning_rate": 8.263636363636366e-06, "num_tokens": 1240781.0, "completions/mean_length": 242.375, "completions/min_length": 208.0, "completions/max_length": 265.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 242.375, "completions/min_terminated_length": 208.0, "completions/max_terminated_length": 265.0, "rewards/meter/mean": 0.7801742553710938, "rewards/meter/std": 0.3070351183414459, "rewards/count_adherence/mean": 0.930555522441864, "rewards/count_adherence/std": 0.05750546231865883, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7266948223114014, "rewards/total_composite/std": 0.2840970754623413, "reward": 0.7266948223114014, "reward_std": 0.2840970754623413, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03169988840818405, "sampling/sampling_logp_difference/max": 2.977978229522705, "sampling/importance_sampling_ratio/min": 0.050895627588033676, "sampling/importance_sampling_ratio/mean": 1.0045069456100464, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18924708105623722, "clip_ratio/low_mean": 0.01109237689524889, "clip_ratio/low_min": 0.01109237689524889, "clip_ratio/high_mean": 0.011733462568372488, "clip_ratio/high_max": 0.011733462568372488, "clip_ratio/region_mean": 0.022825839463621378, "reward_total_mean": 0.7266948223114014, "reward_meter_mean": 0.7801742553710938, "reward_meter_std": 0.3070351183414459, "reward_count_adherence_mean": 0.930555522441864, "reward_count_adherence_std": 0.05750546231865883, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7266948223114014, "reward_total_composite_std": 0.2840970754623413, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 574.0} {"timestamp_utc": "2026-04-11T20:35:19Z", "mode": "train", "global_step": 575, "epoch": 0.02220420142106889, "loss": -0.0409, "grad_norm": 6.091910362243652, "learning_rate": 8.260606060606061e-06, "num_tokens": 1242957.0, "completions/mean_length": 90.0, "completions/min_length": 77.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.0, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.934752345085144, "rewards/meter/std": 0.10161388665437698, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.934752345085144, "rewards/total_composite/std": 0.10161388665437698, "reward": 0.934752345085144, "reward_std": 0.10161387175321579, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07466892898082733, "sampling/sampling_logp_difference/max": 1.3945322036743164, "sampling/importance_sampling_ratio/min": 0.2479490041732788, "sampling/importance_sampling_ratio/mean": 1.009366750717163, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.570578096434474, "clip_ratio/low_mean": 0.017729461658746004, "clip_ratio/low_min": 0.017729461658746004, "clip_ratio/high_mean": 0.05042124609462917, "clip_ratio/high_max": 0.05042124609462917, "clip_ratio/region_mean": 0.06815070775337517, "reward_total_mean": 0.934752345085144, "reward_meter_mean": 0.934752345085144, "reward_meter_std": 0.10161388665437698, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.934752345085144, "reward_total_composite_std": 0.10161388665437698, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 575.0} {"timestamp_utc": "2026-04-11T20:35:23Z", "mode": "train", "global_step": 576, "epoch": 0.022242817423540315, "loss": 0.0045, "grad_norm": 18.79474639892578, "learning_rate": 8.257575757575758e-06, "num_tokens": 1244488.0, "completions/mean_length": 32.375, "completions/min_length": 28.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.375, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6495904922485352, "rewards/meter/std": 0.46892398595809937, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6495904922485352, "rewards/total_composite/std": 0.46892398595809937, "reward": 0.6495904922485352, "reward_std": 0.46892398595809937, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11486772447824478, "sampling/sampling_logp_difference/max": 0.9975643157958984, "sampling/importance_sampling_ratio/min": 0.3687765896320343, "sampling/importance_sampling_ratio/mean": 1.0079195499420166, "sampling/importance_sampling_ratio/max": 1.6851778030395508, "entropy": 1.167568787932396, "clip_ratio/low_mean": 0.03163566067814827, "clip_ratio/low_min": 0.03163566067814827, "clip_ratio/high_mean": 0.08125000167638063, "clip_ratio/high_max": 0.08125000167638063, "clip_ratio/region_mean": 0.1128856623545289, "reward_total_mean": 0.6495904922485352, "reward_meter_mean": 0.6495904922485352, "reward_meter_std": 0.46892398595809937, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6495904922485352, "reward_total_composite_std": 0.46892398595809937, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 576.0} {"timestamp_utc": "2026-04-11T20:35:29Z", "mode": "train", "global_step": 577, "epoch": 0.02228143342601174, "loss": -0.0136, "grad_norm": 4.967631816864014, "learning_rate": 8.254545454545456e-06, "num_tokens": 1246653.0, "completions/mean_length": 94.625, "completions/min_length": 79.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.625, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.5609345436096191, "rewards/meter/std": 0.46121135354042053, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5170981884002686, "rewards/total_composite/std": 0.43609705567359924, "reward": 0.5170981884002686, "reward_std": 0.43609705567359924, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07433033734560013, "sampling/sampling_logp_difference/max": 1.746337890625, "sampling/importance_sampling_ratio/min": 0.17441149055957794, "sampling/importance_sampling_ratio/mean": 1.0093859434127808, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5516252182424068, "clip_ratio/low_mean": 0.030232772696763277, "clip_ratio/low_min": 0.030232772696763277, "clip_ratio/high_mean": 0.02569850441068411, "clip_ratio/high_max": 0.02569850441068411, "clip_ratio/region_mean": 0.055931277107447386, "reward_total_mean": 0.5170981884002686, "reward_meter_mean": 0.5609345436096191, "reward_meter_std": 0.46121135354042053, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5170981884002686, "reward_total_composite_std": 0.43609705567359924, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 577.0} {"timestamp_utc": "2026-04-11T20:35:35Z", "mode": "train", "global_step": 578, "epoch": 0.022320049428483164, "loss": 0.0284, "grad_norm": 3.93625545501709, "learning_rate": 8.251515151515153e-06, "num_tokens": 1249486.0, "completions/mean_length": 164.125, "completions/min_length": 148.0, "completions/max_length": 188.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 164.125, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 188.0, "rewards/meter/mean": 0.9166064262390137, "rewards/meter/std": 0.09252025932073593, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9166064262390137, "rewards/total_composite/std": 0.09252025932073593, "reward": 0.9166064262390137, "reward_std": 0.09252026677131653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07041816413402557, "sampling/sampling_logp_difference/max": 1.4260609149932861, "sampling/importance_sampling_ratio/min": 0.29223909974098206, "sampling/importance_sampling_ratio/mean": 1.0109132528305054, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5197625793516636, "clip_ratio/low_mean": 0.024840134428814054, "clip_ratio/low_min": 0.024840134428814054, "clip_ratio/high_mean": 0.02754599740728736, "clip_ratio/high_max": 0.02754599740728736, "clip_ratio/region_mean": 0.05238613183610141, "reward_total_mean": 0.9166064262390137, "reward_meter_mean": 0.9166064262390137, "reward_meter_std": 0.09252025932073593, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9166064262390137, "reward_total_composite_std": 0.09252025932073593, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 578.0} {"timestamp_utc": "2026-04-11T20:35:40Z", "mode": "train", "global_step": 579, "epoch": 0.022358665430954588, "loss": 0.1901, "grad_norm": 9.262944221496582, "learning_rate": 8.248484848484848e-06, "num_tokens": 1251320.0, "completions/mean_length": 69.25, "completions/min_length": 57.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.25, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9710856676101685, "rewards/meter/std": 0.049965545535087585, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9087815284729004, "rewards/total_composite/std": 0.17285725474357605, "reward": 0.9087815284729004, "reward_std": 0.17285725474357605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.096940778195858, "sampling/sampling_logp_difference/max": 1.1315927505493164, "sampling/importance_sampling_ratio/min": 0.3225191533565521, "sampling/importance_sampling_ratio/mean": 1.0127062797546387, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8376755490899086, "clip_ratio/low_mean": 0.01710526365786791, "clip_ratio/low_min": 0.01710526365786791, "clip_ratio/high_mean": 0.06810012133792043, "clip_ratio/high_max": 0.06810012133792043, "clip_ratio/region_mean": 0.08520538499578834, "reward_total_mean": 0.9087815284729004, "reward_meter_mean": 0.9710856676101685, "reward_meter_std": 0.049965545535087585, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9087815284729004, "reward_total_composite_std": 0.17285725474357605, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 579.0} {"timestamp_utc": "2026-04-11T20:35:50Z", "mode": "train", "global_step": 580, "epoch": 0.022397281433426012, "loss": -0.2893, "grad_norm": 0.5297556519508362, "learning_rate": 8.245454545454546e-06, "num_tokens": 1255245.0, "completions/mean_length": 362.625, "completions/min_length": 316.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 341.2857360839844, "completions/min_terminated_length": 316.0, "completions/max_terminated_length": 367.0, "rewards/meter/mean": 0.9441625475883484, "rewards/meter/std": 0.14516927301883698, "rewards/count_adherence/mean": 0.5588235855102539, "rewards/count_adherence/std": 0.04446641355752945, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.4977211654186249, "rewards/total_composite/std": 0.2027886062860489, "reward": 0.4977211654186249, "reward_std": 0.2027885913848877, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02394985593855381, "sampling/sampling_logp_difference/max": 1.2177810668945312, "sampling/importance_sampling_ratio/min": 0.29588600993156433, "sampling/importance_sampling_ratio/mean": 1.0028082132339478, "sampling/importance_sampling_ratio/max": 1.8625940084457397, "entropy": 0.16224890854209661, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.018500169040635228, "clip_ratio/high_max": 0.018500169040635228, "clip_ratio/region_mean": 0.018500169040635228, "reward_total_mean": 0.4977211654186249, "reward_meter_mean": 0.9441625475883484, "reward_meter_std": 0.14516927301883698, "reward_count_adherence_mean": 0.5588235855102539, "reward_count_adherence_std": 0.04446641355752945, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.4977211654186249, "reward_total_composite_std": 0.2027886062860489, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 580.0} {"timestamp_utc": "2026-04-11T20:35:55Z", "mode": "train", "global_step": 581, "epoch": 0.022435897435897436, "loss": 0.0445, "grad_norm": 6.394820690155029, "learning_rate": 8.242424242424243e-06, "num_tokens": 1257068.0, "completions/mean_length": 67.875, "completions/min_length": 60.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9752446413040161, "rewards/meter/std": 0.03254934400320053, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9752446413040161, "rewards/total_composite/std": 0.03254934400320053, "reward": 0.9752446413040161, "reward_std": 0.03254932910203934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09171347320079803, "sampling/sampling_logp_difference/max": 1.578200101852417, "sampling/importance_sampling_ratio/min": 0.20634615421295166, "sampling/importance_sampling_ratio/mean": 1.0144362449645996, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8226092867553234, "clip_ratio/low_mean": 0.02420289907604456, "clip_ratio/low_min": 0.02420289907604456, "clip_ratio/high_mean": 0.07077483460307121, "clip_ratio/high_max": 0.07077483460307121, "clip_ratio/region_mean": 0.09497773367911577, "reward_total_mean": 0.9752446413040161, "reward_meter_mean": 0.9752446413040161, "reward_meter_std": 0.03254934400320053, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9752446413040161, "reward_total_composite_std": 0.03254934400320053, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 581.0} {"timestamp_utc": "2026-04-11T20:36:00Z", "mode": "train", "global_step": 582, "epoch": 0.02247451343836886, "loss": -0.0091, "grad_norm": 3.4065492153167725, "learning_rate": 8.23939393939394e-06, "num_tokens": 1258849.0, "completions/mean_length": 59.625, "completions/min_length": 56.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.625, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9942880868911743, "rewards/meter/std": 0.002081402810290456, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9942880868911743, "rewards/total_composite/std": 0.002081402810290456, "reward": 0.9942880868911743, "reward_std": 0.0020814072340726852, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04282836988568306, "sampling/sampling_logp_difference/max": 1.182105541229248, "sampling/importance_sampling_ratio/min": 0.3066324293613434, "sampling/importance_sampling_ratio/mean": 1.0069268941879272, "sampling/importance_sampling_ratio/max": 1.7408068180084229, "entropy": 0.32135884277522564, "clip_ratio/low_mean": 0.036390340072102845, "clip_ratio/low_min": 0.036390340072102845, "clip_ratio/high_mean": 0.010697707650251687, "clip_ratio/high_max": 0.010697707650251687, "clip_ratio/region_mean": 0.04708804772235453, "reward_total_mean": 0.9942880868911743, "reward_meter_mean": 0.9942880868911743, "reward_meter_std": 0.002081402810290456, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9942880868911743, "reward_total_composite_std": 0.002081402810290456, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 582.0} {"timestamp_utc": "2026-04-11T20:36:07Z", "mode": "train", "global_step": 583, "epoch": 0.022513129440840284, "loss": 0.0741, "grad_norm": 5.01640510559082, "learning_rate": 8.236363636363637e-06, "num_tokens": 1261386.0, "completions/mean_length": 147.125, "completions/min_length": 129.0, "completions/max_length": 166.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.125, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.7399221062660217, "rewards/meter/std": 0.3321729898452759, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6761964559555054, "rewards/total_composite/std": 0.29238972067832947, "reward": 0.6761964559555054, "reward_std": 0.29238972067832947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06298165768384933, "sampling/sampling_logp_difference/max": 1.7989108562469482, "sampling/importance_sampling_ratio/min": 0.16547901928424835, "sampling/importance_sampling_ratio/mean": 1.0108050107955933, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4634787142276764, "clip_ratio/low_mean": 0.02640543133020401, "clip_ratio/low_min": 0.02640543133020401, "clip_ratio/high_mean": 0.02780228859046474, "clip_ratio/high_max": 0.02780228859046474, "clip_ratio/region_mean": 0.05420771992066875, "reward_total_mean": 0.6761964559555054, "reward_meter_mean": 0.7399221062660217, "reward_meter_std": 0.3321729898452759, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6761964559555054, "reward_total_composite_std": 0.29238972067832947, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 583.0} {"timestamp_utc": "2026-04-11T20:36:12Z", "mode": "train", "global_step": 584, "epoch": 0.02255174544331171, "loss": 0.0272, "grad_norm": 8.98315715789795, "learning_rate": 8.233333333333335e-06, "num_tokens": 1263178.0, "completions/mean_length": 67.0, "completions/min_length": 60.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.7797154188156128, "rewards/meter/std": 0.2954230308532715, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7797154188156128, "rewards/total_composite/std": 0.2954230308532715, "reward": 0.7797154188156128, "reward_std": 0.29542306065559387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10191215574741364, "sampling/sampling_logp_difference/max": 1.1516196727752686, "sampling/importance_sampling_ratio/min": 0.3161243200302124, "sampling/importance_sampling_ratio/mean": 1.0329049825668335, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1980116702616215, "clip_ratio/low_mean": 0.02927643246948719, "clip_ratio/low_min": 0.02927643246948719, "clip_ratio/high_mean": 0.05211255559697747, "clip_ratio/high_max": 0.05211255559697747, "clip_ratio/region_mean": 0.08138898806646466, "reward_total_mean": 0.7797154188156128, "reward_meter_mean": 0.7797154188156128, "reward_meter_std": 0.2954230308532715, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7797154188156128, "reward_total_composite_std": 0.2954230308532715, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 584.0} {"timestamp_utc": "2026-04-11T20:36:19Z", "mode": "train", "global_step": 585, "epoch": 0.022590361445783132, "loss": 0.0176, "grad_norm": 10.492538452148438, "learning_rate": 8.23030303030303e-06, "num_tokens": 1264918.0, "completions/mean_length": 55.5, "completions/min_length": 51.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8827799558639526, "rewards/meter/std": 0.28730660676956177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8827799558639526, "rewards/total_composite/std": 0.28730660676956177, "reward": 0.8827799558639526, "reward_std": 0.28730660676956177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08820638805627823, "sampling/sampling_logp_difference/max": 1.0144014358520508, "sampling/importance_sampling_ratio/min": 0.36261942982673645, "sampling/importance_sampling_ratio/mean": 1.0217286348342896, "sampling/importance_sampling_ratio/max": 1.7023773193359375, "entropy": 0.8991814702749252, "clip_ratio/low_mean": 0.019736841320991516, "clip_ratio/low_min": 0.019736841320991516, "clip_ratio/high_mean": 0.05925852665677667, "clip_ratio/high_max": 0.05925852665677667, "clip_ratio/region_mean": 0.07899536797776818, "reward_total_mean": 0.8827799558639526, "reward_meter_mean": 0.8827799558639526, "reward_meter_std": 0.28730660676956177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8827799558639526, "reward_total_composite_std": 0.28730660676956177, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 585.0} {"timestamp_utc": "2026-04-11T20:36:24Z", "mode": "train", "global_step": 586, "epoch": 0.022628977448254557, "loss": -0.0258, "grad_norm": 5.471312999725342, "learning_rate": 8.227272727272728e-06, "num_tokens": 1266838.0, "completions/mean_length": 67.0, "completions/min_length": 57.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9903906583786011, "rewards/meter/std": 0.01108819991350174, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9903906583786011, "rewards/total_composite/std": 0.01108819991350174, "reward": 0.9903906583786011, "reward_std": 0.011088193394243717, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08893144875764847, "sampling/sampling_logp_difference/max": 0.7949490547180176, "sampling/importance_sampling_ratio/min": 0.4516042470932007, "sampling/importance_sampling_ratio/mean": 1.0199015140533447, "sampling/importance_sampling_ratio/max": 1.8207906484603882, "entropy": 0.9242872335016727, "clip_ratio/low_mean": 0.017296989914029837, "clip_ratio/low_min": 0.017296989914029837, "clip_ratio/high_mean": 0.054413119331002235, "clip_ratio/high_max": 0.054413119331002235, "clip_ratio/region_mean": 0.07171010924503207, "reward_total_mean": 0.9903906583786011, "reward_meter_mean": 0.9903906583786011, "reward_meter_std": 0.01108819991350174, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9903906583786011, "reward_total_composite_std": 0.01108819991350174, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 586.0} {"timestamp_utc": "2026-04-11T20:36:29Z", "mode": "train", "global_step": 587, "epoch": 0.02266759345072598, "loss": 0.0326, "grad_norm": 9.306632995605469, "learning_rate": 8.224242424242425e-06, "num_tokens": 1268600.0, "completions/mean_length": 61.25, "completions/min_length": 55.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.25, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.7887836694717407, "rewards/meter/std": 0.31874561309814453, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7887836694717407, "rewards/total_composite/std": 0.31874561309814453, "reward": 0.7887836694717407, "reward_std": 0.3187456429004669, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10599493235349655, "sampling/sampling_logp_difference/max": 1.4364380836486816, "sampling/importance_sampling_ratio/min": 0.23777316510677338, "sampling/importance_sampling_ratio/mean": 1.0263839960098267, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9658423773944378, "clip_ratio/low_mean": 0.04157197009772062, "clip_ratio/low_min": 0.04157197009772062, "clip_ratio/high_mean": 0.05832049483433366, "clip_ratio/high_max": 0.05832049483433366, "clip_ratio/region_mean": 0.09989246493205428, "reward_total_mean": 0.7887836694717407, "reward_meter_mean": 0.7887836694717407, "reward_meter_std": 0.31874561309814453, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7887836694717407, "reward_total_composite_std": 0.31874561309814453, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 587.0} {"timestamp_utc": "2026-04-11T20:36:39Z", "mode": "train", "global_step": 588, "epoch": 0.022706209453197405, "loss": -0.2794, "grad_norm": 1.4397287368774414, "learning_rate": 8.221212121212122e-06, "num_tokens": 1271261.0, "completions/mean_length": 390.625, "completions/min_length": 229.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 269.25, "completions/min_terminated_length": 229.0, "completions/max_terminated_length": 301.0, "rewards/meter/mean": 0.9679808616638184, "rewards/meter/std": 0.08165019750595093, "rewards/count_adherence/mean": 0.5249999761581421, "rewards/count_adherence/std": 0.2815771996974945, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5204770565032959, "rewards/total_composite/std": 0.28585317730903625, "reward": 0.5204770565032959, "reward_std": 0.28585320711135864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08885669708251953, "sampling/sampling_logp_difference/max": 1.4716987609863281, "sampling/importance_sampling_ratio/min": 0.22953523695468903, "sampling/importance_sampling_ratio/mean": 1.0239464044570923, "sampling/importance_sampling_ratio/max": 1.7755378484725952, "entropy": 0.43688641488552094, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.02804773487150669, "clip_ratio/high_max": 0.02804773487150669, "clip_ratio/region_mean": 0.02804773487150669, "reward_total_mean": 0.5204770565032959, "reward_meter_mean": 0.9679808616638184, "reward_meter_std": 0.08165019750595093, "reward_count_adherence_mean": 0.5249999761581421, "reward_count_adherence_std": 0.2815771996974945, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5204770565032959, "reward_total_composite_std": 0.28585317730903625, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 588.0} {"timestamp_utc": "2026-04-11T20:36:45Z", "mode": "train", "global_step": 589, "epoch": 0.02274482545566883, "loss": -0.0798, "grad_norm": 4.359503269195557, "learning_rate": 8.21818181818182e-06, "num_tokens": 1273507.0, "completions/mean_length": 105.75, "completions/min_length": 88.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.75, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9854410886764526, "rewards/meter/std": 0.019673505797982216, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7688436508178711, "rewards/total_composite/std": 0.07496000826358795, "reward": 0.7688436508178711, "reward_std": 0.07495999336242676, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039897385984659195, "sampling/sampling_logp_difference/max": 1.3246994018554688, "sampling/importance_sampling_ratio/min": 0.2658828794956207, "sampling/importance_sampling_ratio/mean": 1.0047855377197266, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2631940208375454, "clip_ratio/low_mean": 0.03957205289043486, "clip_ratio/low_min": 0.03957205289043486, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.04143772448878735, "reward_total_mean": 0.7688436508178711, "reward_meter_mean": 0.9854410886764526, "reward_meter_std": 0.019673505797982216, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7688436508178711, "reward_total_composite_std": 0.07496000826358795, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 589.0} {"timestamp_utc": "2026-04-11T20:36:55Z", "mode": "train", "global_step": 590, "epoch": 0.022783441458140253, "loss": -0.1856, "grad_norm": 1.4748144149780273, "learning_rate": 8.215151515151517e-06, "num_tokens": 1275902.0, "completions/mean_length": 170.375, "completions/min_length": 101.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 121.5714340209961, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.795690655708313, "rewards/meter/std": 0.37391647696495056, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6351298093795776, "rewards/total_composite/std": 0.378958523273468, "reward": 0.6351298093795776, "reward_std": 0.378958523273468, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03722504526376724, "sampling/sampling_logp_difference/max": 1.0452070236206055, "sampling/importance_sampling_ratio/min": 0.35161903500556946, "sampling/importance_sampling_ratio/mean": 1.005539059638977, "sampling/importance_sampling_ratio/max": 1.574059009552002, "entropy": 0.2874904926866293, "clip_ratio/low_mean": 0.0037128713447600603, "clip_ratio/low_min": 0.0037128713447600603, "clip_ratio/high_mean": 0.029261499643325806, "clip_ratio/high_max": 0.029261499643325806, "clip_ratio/region_mean": 0.032974370988085866, "reward_total_mean": 0.6351298093795776, "reward_meter_mean": 0.795690655708313, "reward_meter_std": 0.37391647696495056, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6351298093795776, "reward_total_composite_std": 0.378958523273468, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 590.0} {"timestamp_utc": "2026-04-11T20:36:59Z", "mode": "train", "global_step": 591, "epoch": 0.022822057460611677, "loss": 0.0618, "grad_norm": 14.161763191223145, "learning_rate": 8.212121212121212e-06, "num_tokens": 1277586.0, "completions/mean_length": 40.5, "completions/min_length": 31.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.5279456377029419, "rewards/meter/std": 0.370597779750824, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5279456377029419, "rewards/total_composite/std": 0.370597779750824, "reward": 0.5279456377029419, "reward_std": 0.370597779750824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17503851652145386, "sampling/sampling_logp_difference/max": 1.1672029495239258, "sampling/importance_sampling_ratio/min": 0.31123626232147217, "sampling/importance_sampling_ratio/mean": 1.0325664281845093, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7381385117769241, "clip_ratio/low_mean": 0.08319734176620841, "clip_ratio/low_min": 0.08319734176620841, "clip_ratio/high_mean": 0.05741901881992817, "clip_ratio/high_max": 0.05741901881992817, "clip_ratio/region_mean": 0.14061636058613658, "reward_total_mean": 0.5279456377029419, "reward_meter_mean": 0.5279456377029419, "reward_meter_std": 0.370597779750824, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5279456377029419, "reward_total_composite_std": 0.370597779750824, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 591.0} {"timestamp_utc": "2026-04-11T20:37:09Z", "mode": "train", "global_step": 592, "epoch": 0.0228606734630831, "loss": -0.0262, "grad_norm": 1.401776909828186, "learning_rate": 8.20909090909091e-06, "num_tokens": 1282466.0, "completions/mean_length": 389.0, "completions/min_length": 360.0, "completions/max_length": 428.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 389.0, "completions/min_terminated_length": 360.0, "completions/max_terminated_length": 428.0, "rewards/meter/mean": 0.9966329336166382, "rewards/meter/std": 0.0009374520741403103, "rewards/count_adherence/mean": 0.5723684430122375, "rewards/count_adherence/std": 0.05215953662991524, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5704156160354614, "rewards/total_composite/std": 0.05163882300257683, "reward": 0.5704156160354614, "reward_std": 0.05163882300257683, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017461512237787247, "sampling/sampling_logp_difference/max": 1.5229759216308594, "sampling/importance_sampling_ratio/min": 0.21806199848651886, "sampling/importance_sampling_ratio/mean": 1.001259446144104, "sampling/importance_sampling_ratio/max": 1.6324708461761475, "entropy": 0.12590192956849933, "clip_ratio/low_mean": 0.0036123525351285934, "clip_ratio/low_min": 0.0036123525351285934, "clip_ratio/high_mean": 0.010578885208815336, "clip_ratio/high_max": 0.010578885208815336, "clip_ratio/region_mean": 0.01419123774394393, "reward_total_mean": 0.5704156160354614, "reward_meter_mean": 0.9966329336166382, "reward_meter_std": 0.0009374520741403103, "reward_count_adherence_mean": 0.5723684430122375, "reward_count_adherence_std": 0.05215953662991524, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5704156160354614, "reward_total_composite_std": 0.05163882300257683, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 592.0} {"timestamp_utc": "2026-04-11T20:37:13Z", "mode": "train", "global_step": 593, "epoch": 0.022899289465554525, "loss": 0.0288, "grad_norm": 7.278051853179932, "learning_rate": 8.206060606060607e-06, "num_tokens": 1284101.0, "completions/mean_length": 62.375, "completions/min_length": 56.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8718899488449097, "rewards/meter/std": 0.3202956020832062, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8718899488449097, "rewards/total_composite/std": 0.3202956020832062, "reward": 0.8718899488449097, "reward_std": 0.3202956020832062, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06579845398664474, "sampling/sampling_logp_difference/max": 1.4545068740844727, "sampling/importance_sampling_ratio/min": 0.23351548612117767, "sampling/importance_sampling_ratio/mean": 1.0134860277175903, "sampling/importance_sampling_ratio/max": 1.7530381679534912, "entropy": 0.5911059454083443, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/high_mean": 0.06137674581259489, "clip_ratio/high_max": 0.06137674581259489, "clip_ratio/region_mean": 0.06522289966233075, "reward_total_mean": 0.8718899488449097, "reward_meter_mean": 0.8718899488449097, "reward_meter_std": 0.3202956020832062, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8718899488449097, "reward_total_composite_std": 0.3202956020832062, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 593.0} {"timestamp_utc": "2026-04-11T20:37:20Z", "mode": "train", "global_step": 594, "epoch": 0.02293790546802595, "loss": -0.0155, "grad_norm": 4.129243850708008, "learning_rate": 8.203030303030304e-06, "num_tokens": 1287579.0, "completions/mean_length": 221.75, "completions/min_length": 195.0, "completions/max_length": 234.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 221.75, "completions/min_terminated_length": 195.0, "completions/max_terminated_length": 234.0, "rewards/meter/mean": 0.9475843906402588, "rewards/meter/std": 0.13766330480575562, "rewards/count_adherence/mean": 0.8928571939468384, "rewards/count_adherence/std": 0.06613000482320786, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7827649116516113, "rewards/total_composite/std": 0.32273051142692566, "reward": 0.7827649116516113, "reward_std": 0.32273048162460327, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05360095202922821, "sampling/sampling_logp_difference/max": 1.5283584594726562, "sampling/importance_sampling_ratio/min": 0.21689140796661377, "sampling/importance_sampling_ratio/mean": 1.0077857971191406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49583121575415134, "clip_ratio/low_mean": 0.004878048785030842, "clip_ratio/low_min": 0.004878048785030842, "clip_ratio/high_mean": 0.029324828181415796, "clip_ratio/high_max": 0.029324828181415796, "clip_ratio/region_mean": 0.03420287696644664, "reward_total_mean": 0.7827649116516113, "reward_meter_mean": 0.9475843906402588, "reward_meter_std": 0.13766330480575562, "reward_count_adherence_mean": 0.8928571939468384, "reward_count_adherence_std": 0.06613000482320786, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7827649116516113, "reward_total_composite_std": 0.32273051142692566, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 594.0} {"timestamp_utc": "2026-04-11T20:37:25Z", "mode": "train", "global_step": 595, "epoch": 0.022976521470497373, "loss": -0.0029, "grad_norm": 5.483264446258545, "learning_rate": 8.2e-06, "num_tokens": 1289294.0, "completions/mean_length": 69.375, "completions/min_length": 59.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.375, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.825728178024292, "rewards/meter/std": 0.15038101375102997, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.825728178024292, "rewards/total_composite/std": 0.15038101375102997, "reward": 0.825728178024292, "reward_std": 0.15038102865219116, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.043961603194475174, "sampling/sampling_logp_difference/max": 2.089282989501953, "sampling/importance_sampling_ratio/min": 0.12377584725618362, "sampling/importance_sampling_ratio/mean": 1.0042104721069336, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2160019911825657, "clip_ratio/low_mean": 0.012580781243741512, "clip_ratio/low_min": 0.012580781243741512, "clip_ratio/high_mean": 0.014449786627665162, "clip_ratio/high_max": 0.014449786627665162, "clip_ratio/region_mean": 0.027030567871406674, "reward_total_mean": 0.825728178024292, "reward_meter_mean": 0.825728178024292, "reward_meter_std": 0.15038101375102997, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.825728178024292, "reward_total_composite_std": 0.15038101375102997, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 595.0} {"timestamp_utc": "2026-04-11T20:37:31Z", "mode": "train", "global_step": 596, "epoch": 0.023015137472968798, "loss": 0.0213, "grad_norm": 4.874330520629883, "learning_rate": 8.196969696969698e-06, "num_tokens": 1291573.0, "completions/mean_length": 119.875, "completions/min_length": 114.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.875, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.9766520857810974, "rewards/meter/std": 0.04947783052921295, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9766520857810974, "rewards/total_composite/std": 0.04947783052921295, "reward": 0.9766520857810974, "reward_std": 0.04947783797979355, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02237871289253235, "sampling/sampling_logp_difference/max": 1.3679141998291016, "sampling/importance_sampling_ratio/min": 0.25463753938674927, "sampling/importance_sampling_ratio/mean": 1.0042794942855835, "sampling/importance_sampling_ratio/max": 1.7739815711975098, "entropy": 0.1262477389536798, "clip_ratio/low_mean": 0.005905511789023876, "clip_ratio/low_min": 0.005905511789023876, "clip_ratio/high_mean": 0.015763025381602347, "clip_ratio/high_max": 0.015763025381602347, "clip_ratio/region_mean": 0.021668537170626223, "reward_total_mean": 0.9766520857810974, "reward_meter_mean": 0.9766520857810974, "reward_meter_std": 0.04947783052921295, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9766520857810974, "reward_total_composite_std": 0.04947783052921295, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 596.0} {"timestamp_utc": "2026-04-11T20:37:36Z", "mode": "train", "global_step": 597, "epoch": 0.02305375347544022, "loss": 0.0304, "grad_norm": 10.373922348022461, "learning_rate": 8.193939393939394e-06, "num_tokens": 1293343.0, "completions/mean_length": 62.25, "completions/min_length": 57.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.25, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.7612351775169373, "rewards/meter/std": 0.35920435190200806, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7612351775169373, "rewards/total_composite/std": 0.35920435190200806, "reward": 0.7612351775169373, "reward_std": 0.35920435190200806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0720410943031311, "sampling/sampling_logp_difference/max": 1.0596399307250977, "sampling/importance_sampling_ratio/min": 0.3465805947780609, "sampling/importance_sampling_ratio/mean": 1.0154905319213867, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4781202729791403, "clip_ratio/low_mean": 0.01575682358816266, "clip_ratio/low_min": 0.01575682358816266, "clip_ratio/high_mean": 0.037429331336170435, "clip_ratio/high_max": 0.037429331336170435, "clip_ratio/region_mean": 0.053186154924333096, "reward_total_mean": 0.7612351775169373, "reward_meter_mean": 0.7612351775169373, "reward_meter_std": 0.35920435190200806, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7612351775169373, "reward_total_composite_std": 0.35920435190200806, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 597.0} {"timestamp_utc": "2026-04-11T20:37:40Z", "mode": "train", "global_step": 598, "epoch": 0.023092369477911646, "loss": 0.0542, "grad_norm": 11.545695304870605, "learning_rate": 8.190909090909091e-06, "num_tokens": 1294775.0, "completions/mean_length": 29.0, "completions/min_length": 27.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9914101958274841, "rewards/meter/std": 0.00670025497674942, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9914101958274841, "rewards/total_composite/std": 0.00670025497674942, "reward": 0.9914101958274841, "reward_std": 0.006700248457491398, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06982678174972534, "sampling/sampling_logp_difference/max": 1.5482743978500366, "sampling/importance_sampling_ratio/min": 0.2126145362854004, "sampling/importance_sampling_ratio/mean": 1.004575490951538, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44570457749068737, "clip_ratio/low_mean": 0.027435938362032175, "clip_ratio/low_min": 0.027435938362032175, "clip_ratio/high_mean": 0.02251984179019928, "clip_ratio/high_max": 0.02251984179019928, "clip_ratio/region_mean": 0.049955780152231455, "reward_total_mean": 0.9914101958274841, "reward_meter_mean": 0.9914101958274841, "reward_meter_std": 0.00670025497674942, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9914101958274841, "reward_total_composite_std": 0.00670025497674942, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 598.0} {"timestamp_utc": "2026-04-11T20:37:45Z", "mode": "train", "global_step": 599, "epoch": 0.02313098548038307, "loss": 0.0205, "grad_norm": 11.035534858703613, "learning_rate": 8.187878787878788e-06, "num_tokens": 1296492.0, "completions/mean_length": 59.625, "completions/min_length": 52.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.625, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.7003433704376221, "rewards/meter/std": 0.4035126864910126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7003433704376221, "rewards/total_composite/std": 0.4035126864910126, "reward": 0.7003433704376221, "reward_std": 0.4035126864910126, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09578597545623779, "sampling/sampling_logp_difference/max": 1.493178367614746, "sampling/importance_sampling_ratio/min": 0.22465747594833374, "sampling/importance_sampling_ratio/mean": 1.020950198173523, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6933459341526031, "clip_ratio/low_mean": 0.056075175292789936, "clip_ratio/low_min": 0.056075175292789936, "clip_ratio/high_mean": 0.04812374617904425, "clip_ratio/high_max": 0.04812374617904425, "clip_ratio/region_mean": 0.10419892147183418, "reward_total_mean": 0.7003433704376221, "reward_meter_mean": 0.7003433704376221, "reward_meter_std": 0.4035126864910126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7003433704376221, "reward_total_composite_std": 0.4035126864910126, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 599.0} {"timestamp_utc": "2026-04-11T20:37:55Z", "mode": "train", "global_step": 600, "epoch": 0.023169601482854494, "loss": -0.1774, "grad_norm": 1.266343355178833, "learning_rate": 8.184848484848486e-06, "num_tokens": 1298416.0, "completions/mean_length": 128.5, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.71428680419922, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9104650616645813, "rewards/meter/std": 0.24043236672878265, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8710224628448486, "rewards/total_composite/std": 0.3519780933856964, "reward": 0.8710224628448486, "reward_std": 0.3519781231880188, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07025135308504105, "sampling/sampling_logp_difference/max": 0.9205818176269531, "sampling/importance_sampling_ratio/min": 0.39828726649284363, "sampling/importance_sampling_ratio/mean": 1.0156185626983643, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5339640863239765, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.044667589012533426, "clip_ratio/high_max": 0.044667589012533426, "clip_ratio/region_mean": 0.044667589012533426, "reward_total_mean": 0.8710224628448486, "reward_meter_mean": 0.9104650616645813, "reward_meter_std": 0.24043236672878265, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8710224628448486, "reward_total_composite_std": 0.3519780933856964, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 600.0} {"timestamp_utc": "2026-04-11T20:39:08Z", "mode": "eval", "global_step": 600, "epoch": 0.023169601482854494, "eval_loss": NaN, "eval_runtime": 72.8099, "eval_samples_per_second": 1.428, "eval_steps_per_second": 0.179, "eval_num_tokens": 1298416.0, "eval_completions/mean_length": 205.16346153846155, "eval_completions/min_length": 58.92307692307692, "eval_completions/max_length": 382.61538461538464, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/mean_terminated_length": 186.50274892953726, "eval_completions/min_terminated_length": 58.92307692307692, "eval_completions/max_terminated_length": 339.2307692307692, "eval_rewards/meter/mean": 0.6951331358689529, "eval_rewards/meter/std": 0.4035135255410121, "eval_rewards/count_adherence/mean": 0.9103590066616352, "eval_rewards/count_adherence/std": 0.12672624450463515, "eval_rewards/arabic_clean/mean": 0.9326923076923077, "eval_rewards/arabic_clean/std": 0.13402181176038888, "eval_rewards/total_composite/mean": 0.6174316452099726, "eval_rewards/total_composite/std": 0.3925590217113495, "eval_reward": 0.6174316452099726, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.023065941838117745, "eval_sampling/sampling_logp_difference/max": 0.9544231708233173, "eval_sampling/importance_sampling_ratio/min": 0.39233631583360523, "eval_sampling/importance_sampling_ratio/mean": 1.0067635774612427, "eval_sampling/importance_sampling_ratio/max": 1.4073029848245473, "eval_entropy": 0.2500755110612282, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6174316452099726, "eval_reward_meter_mean": 0.6951331358689529, "eval_reward_meter_std": 0.4035135255410121, "eval_reward_count_adherence_mean": 0.9103590066616352, "eval_reward_count_adherence_std": 0.12672624450463515, "eval_reward_arabic_clean_mean": 0.9326923076923077, "eval_reward_arabic_clean_std": 0.13402181176038888, "eval_reward_total_composite_mean": 0.6174316452099726, "eval_reward_total_composite_std": 0.3925590217113495, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 600.0} {"timestamp_utc": "2026-04-11T20:39:15Z", "mode": "train", "global_step": 601, "epoch": 0.023208217485325918, "loss": -0.0232, "grad_norm": 4.212752342224121, "learning_rate": 8.181818181818183e-06, "num_tokens": 1300265.0, "completions/mean_length": 59.125, "completions/min_length": 57.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9931149482727051, "rewards/meter/std": 0.003802346298471093, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9931149482727051, "rewards/total_composite/std": 0.003802346298471093, "reward": 0.9931149482727051, "reward_std": 0.0038023567758500576, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03990501910448074, "sampling/sampling_logp_difference/max": 1.0637364387512207, "sampling/importance_sampling_ratio/min": 0.3451637029647827, "sampling/importance_sampling_ratio/mean": 1.0037661790847778, "sampling/importance_sampling_ratio/max": 1.6298530101776123, "entropy": 0.31142993830144405, "clip_ratio/low_mean": 0.015199637040495872, "clip_ratio/low_min": 0.015199637040495872, "clip_ratio/high_mean": 0.024640035582706332, "clip_ratio/high_max": 0.024640035582706332, "clip_ratio/region_mean": 0.039839672623202205, "reward_total_mean": 0.9931149482727051, "reward_meter_mean": 0.9931149482727051, "reward_meter_std": 0.003802346298471093, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9931149482727051, "reward_total_composite_std": 0.003802346298471093, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 601.0} {"timestamp_utc": "2026-04-11T20:39:25Z", "mode": "train", "global_step": 602, "epoch": 0.023246833487797342, "loss": -0.0699, "grad_norm": 4.308811664581299, "learning_rate": 8.17878787878788e-06, "num_tokens": 1302514.0, "completions/mean_length": 157.125, "completions/min_length": 89.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 106.42857360839844, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.803666353225708, "rewards/meter/std": 0.3494153916835785, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7166256308555603, "rewards/total_composite/std": 0.3973376154899597, "reward": 0.7166256308555603, "reward_std": 0.3973376154899597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05126181244850159, "sampling/sampling_logp_difference/max": 1.8974881172180176, "sampling/importance_sampling_ratio/min": 0.14994479715824127, "sampling/importance_sampling_ratio/mean": 1.0027865171432495, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3072041980922222, "clip_ratio/low_mean": 0.019871795549988747, "clip_ratio/low_min": 0.019871795549988747, "clip_ratio/high_mean": 0.029445810709148645, "clip_ratio/high_max": 0.029445810709148645, "clip_ratio/region_mean": 0.04931760625913739, "reward_total_mean": 0.7166256308555603, "reward_meter_mean": 0.803666353225708, "reward_meter_std": 0.3494153916835785, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7166256308555603, "reward_total_composite_std": 0.3973376154899597, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 602.0} {"timestamp_utc": "2026-04-11T20:39:35Z", "mode": "train", "global_step": 603, "epoch": 0.023285449490268766, "loss": -0.1352, "grad_norm": 3.3487203121185303, "learning_rate": 8.175757575757577e-06, "num_tokens": 1304712.0, "completions/mean_length": 165.75, "completions/min_length": 110.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 116.28572082519531, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.7495772838592529, "rewards/meter/std": 0.4524652659893036, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7459595203399658, "rewards/total_composite/std": 0.45911717414855957, "reward": 0.7459595203399658, "reward_std": 0.45911717414855957, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034356024116277695, "sampling/sampling_logp_difference/max": 1.8989930152893066, "sampling/importance_sampling_ratio/min": 0.14971929788589478, "sampling/importance_sampling_ratio/mean": 0.9969672560691833, "sampling/importance_sampling_ratio/max": 1.7410459518432617, "entropy": 0.19555421639233828, "clip_ratio/low_mean": 0.006521739065647125, "clip_ratio/low_min": 0.006521739065647125, "clip_ratio/high_mean": 0.027199887670576572, "clip_ratio/high_max": 0.027199887670576572, "clip_ratio/region_mean": 0.0337216267362237, "reward_total_mean": 0.7459595203399658, "reward_meter_mean": 0.7495772838592529, "reward_meter_std": 0.4524652659893036, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7459595203399658, "reward_total_composite_std": 0.45911717414855957, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 603.0} {"timestamp_utc": "2026-04-11T20:39:39Z", "mode": "train", "global_step": 604, "epoch": 0.02332406549274019, "loss": 0.0446, "grad_norm": 9.974048614501953, "learning_rate": 8.172727272727273e-06, "num_tokens": 1306447.0, "completions/mean_length": 56.875, "completions/min_length": 53.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8660935163497925, "rewards/meter/std": 0.2912440299987793, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8660935163497925, "rewards/total_composite/std": 0.2912440299987793, "reward": 0.8660935163497925, "reward_std": 0.2912440299987793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09364975243806839, "sampling/sampling_logp_difference/max": 1.1022062301635742, "sampling/importance_sampling_ratio/min": 0.33213749527931213, "sampling/importance_sampling_ratio/mean": 1.023914098739624, "sampling/importance_sampling_ratio/max": 1.9299688339233398, "entropy": 0.8086119182407856, "clip_ratio/low_mean": 0.014390989672392607, "clip_ratio/low_min": 0.014390989672392607, "clip_ratio/high_mean": 0.05441663879901171, "clip_ratio/high_max": 0.05441663879901171, "clip_ratio/region_mean": 0.06880762847140431, "reward_total_mean": 0.8660935163497925, "reward_meter_mean": 0.8660935163497925, "reward_meter_std": 0.2912440299987793, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8660935163497925, "reward_total_composite_std": 0.2912440299987793, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 604.0} {"timestamp_utc": "2026-04-11T20:39:49Z", "mode": "train", "global_step": 605, "epoch": 0.023362681495211614, "loss": -0.2736, "grad_norm": 0.45873093605041504, "learning_rate": 8.16969696969697e-06, "num_tokens": 1309849.0, "completions/mean_length": 282.25, "completions/min_length": 239.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 249.4285888671875, "completions/min_terminated_length": 239.0, "completions/max_terminated_length": 256.0, "rewards/meter/mean": 0.9013054370880127, "rewards/meter/std": 0.2691211402416229, "rewards/count_adherence/mean": 0.9107142686843872, "rewards/count_adherence/std": 0.25253814458847046, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8718953132629395, "rewards/total_composite/std": 0.352304071187973, "reward": 0.8718953132629395, "reward_std": 0.352304071187973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01991911418735981, "sampling/sampling_logp_difference/max": 1.0003671646118164, "sampling/importance_sampling_ratio/min": 0.36774441599845886, "sampling/importance_sampling_ratio/mean": 1.0060789585113525, "sampling/importance_sampling_ratio/max": 1.9584522247314453, "entropy": 0.12214899715036154, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.008479945652652532, "clip_ratio/high_max": 0.008479945652652532, "clip_ratio/region_mean": 0.008479945652652532, "reward_total_mean": 0.8718953132629395, "reward_meter_mean": 0.9013054370880127, "reward_meter_std": 0.2691211402416229, "reward_count_adherence_mean": 0.9107142686843872, "reward_count_adherence_std": 0.25253814458847046, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8718953132629395, "reward_total_composite_std": 0.352304071187973, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 605.0} {"timestamp_utc": "2026-04-11T20:39:59Z", "mode": "train", "global_step": 606, "epoch": 0.02340129749768304, "loss": -0.2859, "grad_norm": 0.30050450563430786, "learning_rate": 8.166666666666668e-06, "num_tokens": 1313978.0, "completions/mean_length": 359.125, "completions/min_length": 323.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 337.2857360839844, "completions/min_terminated_length": 323.0, "completions/max_terminated_length": 343.0, "rewards/meter/mean": 0.995347797870636, "rewards/meter/std": 0.005631509702652693, "rewards/count_adherence/mean": 0.644230842590332, "rewards/count_adherence/std": 0.23236627876758575, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6420051455497742, "rewards/total_composite/std": 0.23203761875629425, "reward": 0.6420051455497742, "reward_std": 0.23203760385513306, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009540513157844543, "sampling/sampling_logp_difference/max": 0.9852609634399414, "sampling/importance_sampling_ratio/min": 0.37334179878234863, "sampling/importance_sampling_ratio/mean": 1.0007063150405884, "sampling/importance_sampling_ratio/max": 1.6662840843200684, "entropy": 0.061391755007207394, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007427771924994886, "clip_ratio/high_max": 0.007427771924994886, "clip_ratio/region_mean": 0.007427771924994886, "reward_total_mean": 0.6420051455497742, "reward_meter_mean": 0.995347797870636, "reward_meter_std": 0.005631509702652693, "reward_count_adherence_mean": 0.644230842590332, "reward_count_adherence_std": 0.23236627876758575, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6420051455497742, "reward_total_composite_std": 0.23203761875629425, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 606.0} {"timestamp_utc": "2026-04-11T20:40:06Z", "mode": "train", "global_step": 607, "epoch": 0.023439913500154463, "loss": 0.0658, "grad_norm": 3.355532169342041, "learning_rate": 8.163636363636365e-06, "num_tokens": 1316671.0, "completions/mean_length": 154.625, "completions/min_length": 135.0, "completions/max_length": 175.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.625, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 175.0, "rewards/meter/mean": 0.4418049454689026, "rewards/meter/std": 0.4134439527988434, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4215872883796692, "rewards/total_composite/std": 0.38700953125953674, "reward": 0.4215872883796692, "reward_std": 0.38700950145721436, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03203287720680237, "sampling/sampling_logp_difference/max": 1.925581455230713, "sampling/importance_sampling_ratio/min": 0.1457909643650055, "sampling/importance_sampling_ratio/mean": 1.0042226314544678, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1557165440171957, "clip_ratio/low_mean": 0.01970675610937178, "clip_ratio/low_min": 0.01970675610937178, "clip_ratio/high_mean": 0.00936650182120502, "clip_ratio/high_max": 0.00936650182120502, "clip_ratio/region_mean": 0.0290732579305768, "reward_total_mean": 0.4215872883796692, "reward_meter_mean": 0.4418049454689026, "reward_meter_std": 0.4134439527988434, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4215872883796692, "reward_total_composite_std": 0.38700953125953674, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 607.0} {"timestamp_utc": "2026-04-11T20:40:10Z", "mode": "train", "global_step": 608, "epoch": 0.023478529502625887, "loss": -0.0142, "grad_norm": 6.676723957061768, "learning_rate": 8.16060606060606e-06, "num_tokens": 1318477.0, "completions/mean_length": 64.75, "completions/min_length": 59.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.75, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.6355655789375305, "rewards/meter/std": 0.38601288199424744, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6355655789375305, "rewards/total_composite/std": 0.38601288199424744, "reward": 0.6355655789375305, "reward_std": 0.38601288199424744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03938113898038864, "sampling/sampling_logp_difference/max": 1.4897899627685547, "sampling/importance_sampling_ratio/min": 0.2254199981689453, "sampling/importance_sampling_ratio/mean": 1.007306694984436, "sampling/importance_sampling_ratio/max": 1.405138373374939, "entropy": 0.30151631124317646, "clip_ratio/low_mean": 0.013556188903748989, "clip_ratio/low_min": 0.013556188903748989, "clip_ratio/high_mean": 0.011494944454170763, "clip_ratio/high_max": 0.011494944454170763, "clip_ratio/region_mean": 0.025051133357919753, "reward_total_mean": 0.6355655789375305, "reward_meter_mean": 0.6355655789375305, "reward_meter_std": 0.38601288199424744, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6355655789375305, "reward_total_composite_std": 0.38601288199424744, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 608.0} {"timestamp_utc": "2026-04-11T20:40:15Z", "mode": "train", "global_step": 609, "epoch": 0.02351714550509731, "loss": 0.0344, "grad_norm": 11.4707612991333, "learning_rate": 8.15757575757576e-06, "num_tokens": 1320007.0, "completions/mean_length": 32.25, "completions/min_length": 29.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9407675266265869, "rewards/meter/std": 0.10082443803548813, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9407675266265869, "rewards/total_composite/std": 0.10082443803548813, "reward": 0.9407675266265869, "reward_std": 0.10082443803548813, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09047635644674301, "sampling/sampling_logp_difference/max": 1.7924613952636719, "sampling/importance_sampling_ratio/min": 0.16654972732067108, "sampling/importance_sampling_ratio/mean": 1.0182549953460693, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6633263751864433, "clip_ratio/low_mean": 0.03448660718277097, "clip_ratio/low_min": 0.03448660718277097, "clip_ratio/high_mean": 0.08979849983006716, "clip_ratio/high_max": 0.08979849983006716, "clip_ratio/region_mean": 0.12428510701283813, "reward_total_mean": 0.9407675266265869, "reward_meter_mean": 0.9407675266265869, "reward_meter_std": 0.10082443803548813, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9407675266265869, "reward_total_composite_std": 0.10082443803548813, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 609.0} {"timestamp_utc": "2026-04-11T20:40:25Z", "mode": "train", "global_step": 610, "epoch": 0.023555761507568735, "loss": -0.0688, "grad_norm": 3.2543447017669678, "learning_rate": 8.154545454545455e-06, "num_tokens": 1321653.0, "completions/mean_length": 123.75, "completions/min_length": 61.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 68.28572082519531, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.5480015277862549, "rewards/meter/std": 0.4364262819290161, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.547182559967041, "rewards/total_composite/std": 0.43759214878082275, "reward": 0.547182559967041, "reward_std": 0.43759211897850037, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08188275992870331, "sampling/sampling_logp_difference/max": 1.111203670501709, "sampling/importance_sampling_ratio/min": 0.32916250824928284, "sampling/importance_sampling_ratio/mean": 1.0165828466415405, "sampling/importance_sampling_ratio/max": 1.6886634826660156, "entropy": 0.6228674612939358, "clip_ratio/low_mean": 0.019085082225501537, "clip_ratio/low_min": 0.019085082225501537, "clip_ratio/high_mean": 0.023026442737318575, "clip_ratio/high_max": 0.023026442737318575, "clip_ratio/region_mean": 0.04211152496282011, "reward_total_mean": 0.547182559967041, "reward_meter_mean": 0.5480015277862549, "reward_meter_std": 0.4364262819290161, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.547182559967041, "reward_total_composite_std": 0.43759214878082275, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 610.0} {"timestamp_utc": "2026-04-11T20:40:29Z", "mode": "train", "global_step": 611, "epoch": 0.02359437751004016, "loss": -0.011, "grad_norm": 6.743978023529053, "learning_rate": 8.151515151515152e-06, "num_tokens": 1323107.0, "completions/mean_length": 30.75, "completions/min_length": 28.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9942895174026489, "rewards/meter/std": 0.001669783261604607, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9942895174026489, "rewards/total_composite/std": 0.001669783261604607, "reward": 0.9942895174026489, "reward_std": 0.001669783960096538, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05331292748451233, "sampling/sampling_logp_difference/max": 1.7752132415771484, "sampling/importance_sampling_ratio/min": 0.16944730281829834, "sampling/importance_sampling_ratio/mean": 1.0020709037780762, "sampling/importance_sampling_ratio/max": 1.3152281045913696, "entropy": 0.3008997645229101, "clip_ratio/low_mean": 0.012284422758966684, "clip_ratio/low_min": 0.012284422758966684, "clip_ratio/high_mean": 0.01991767482832074, "clip_ratio/high_max": 0.01991767482832074, "clip_ratio/region_mean": 0.032202097587287426, "reward_total_mean": 0.9942895174026489, "reward_meter_mean": 0.9942895174026489, "reward_meter_std": 0.001669783261604607, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9942895174026489, "reward_total_composite_std": 0.001669783261604607, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 611.0} {"timestamp_utc": "2026-04-11T20:40:35Z", "mode": "train", "global_step": 612, "epoch": 0.023632993512511583, "loss": 0.0336, "grad_norm": 7.670464515686035, "learning_rate": 8.14848484848485e-06, "num_tokens": 1324792.0, "completions/mean_length": 55.625, "completions/min_length": 51.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8599371910095215, "rewards/meter/std": 0.3198857009410858, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8599371910095215, "rewards/total_composite/std": 0.3198857009410858, "reward": 0.8599371910095215, "reward_std": 0.31988564133644104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06448192894458771, "sampling/sampling_logp_difference/max": 0.9435205459594727, "sampling/importance_sampling_ratio/min": 0.41206616163253784, "sampling/importance_sampling_ratio/mean": 1.0190188884735107, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5617792904376984, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/high_mean": 0.038389022229239345, "clip_ratio/high_max": 0.038389022229239345, "clip_ratio/region_mean": 0.04672235599718988, "reward_total_mean": 0.8599371910095215, "reward_meter_mean": 0.8599371910095215, "reward_meter_std": 0.3198857009410858, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8599371910095215, "reward_total_composite_std": 0.3198857009410858, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 612.0} {"timestamp_utc": "2026-04-11T20:40:44Z", "mode": "train", "global_step": 613, "epoch": 0.023671609514983007, "loss": -0.087, "grad_norm": 3.4569411277770996, "learning_rate": 8.145454545454547e-06, "num_tokens": 1326353.0, "completions/mean_length": 114.125, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.28571701049805, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.6199837923049927, "rewards/meter/std": 0.4558931887149811, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6004748940467834, "rewards/total_composite/std": 0.48121726512908936, "reward": 0.6004748940467834, "reward_std": 0.48121726512908936, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11160858720541, "sampling/sampling_logp_difference/max": 1.69767427444458, "sampling/importance_sampling_ratio/min": 0.18310889601707458, "sampling/importance_sampling_ratio/mean": 1.0191153287887573, "sampling/importance_sampling_ratio/max": 1.9146498441696167, "entropy": 1.4391320049762726, "clip_ratio/low_mean": 0.02302631549537182, "clip_ratio/low_min": 0.02302631549537182, "clip_ratio/high_mean": 0.05149728851392865, "clip_ratio/high_max": 0.05149728851392865, "clip_ratio/region_mean": 0.07452360400930047, "reward_total_mean": 0.6004748940467834, "reward_meter_mean": 0.6199837923049927, "reward_meter_std": 0.4558931887149811, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6004748940467834, "reward_total_composite_std": 0.48121726512908936, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 613.0} {"timestamp_utc": "2026-04-11T20:40:50Z", "mode": "train", "global_step": 614, "epoch": 0.02371022551745443, "loss": -0.025, "grad_norm": 5.5195722579956055, "learning_rate": 8.142424242424242e-06, "num_tokens": 1328678.0, "completions/mean_length": 122.625, "completions/min_length": 95.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.625, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.5231146216392517, "rewards/meter/std": 0.36290431022644043, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.48351413011550903, "rewards/total_composite/std": 0.3324858844280243, "reward": 0.48351413011550903, "reward_std": 0.3324859142303467, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05160455033183098, "sampling/sampling_logp_difference/max": 3.038996458053589, "sampling/importance_sampling_ratio/min": 0.047882918268442154, "sampling/importance_sampling_ratio/mean": 1.0053133964538574, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.29260148853063583, "clip_ratio/low_mean": 0.022413392609450966, "clip_ratio/low_min": 0.022413392609450966, "clip_ratio/high_mean": 0.02200371865183115, "clip_ratio/high_max": 0.02200371865183115, "clip_ratio/region_mean": 0.044417111261282116, "reward_total_mean": 0.48351413011550903, "reward_meter_mean": 0.5231146216392517, "reward_meter_std": 0.36290431022644043, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.48351413011550903, "reward_total_composite_std": 0.3324858844280243, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 614.0} {"timestamp_utc": "2026-04-11T20:40:55Z", "mode": "train", "global_step": 615, "epoch": 0.023748841519925856, "loss": 0.016, "grad_norm": 4.773278713226318, "learning_rate": 8.139393939393941e-06, "num_tokens": 1330808.0, "completions/mean_length": 105.25, "completions/min_length": 98.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.25, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.7684873342514038, "rewards/meter/std": 0.3484794795513153, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7684873342514038, "rewards/total_composite/std": 0.3484794795513153, "reward": 0.7684873342514038, "reward_std": 0.3484795093536377, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.054694000631570816, "sampling/sampling_logp_difference/max": 1.171186923980713, "sampling/importance_sampling_ratio/min": 0.3099987804889679, "sampling/importance_sampling_ratio/mean": 1.013346791267395, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42340752109885216, "clip_ratio/low_mean": 0.014563622185960412, "clip_ratio/low_min": 0.014563622185960412, "clip_ratio/high_mean": 0.018577482085675, "clip_ratio/high_max": 0.018577482085675, "clip_ratio/region_mean": 0.03314110427163541, "reward_total_mean": 0.7684873342514038, "reward_meter_mean": 0.7684873342514038, "reward_meter_std": 0.3484794795513153, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7684873342514038, "reward_total_composite_std": 0.3484794795513153, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 615.0} {"timestamp_utc": "2026-04-11T20:41:02Z", "mode": "train", "global_step": 616, "epoch": 0.023787457522397283, "loss": 0.0206, "grad_norm": 2.1530308723449707, "learning_rate": 8.136363636363637e-06, "num_tokens": 1333982.0, "completions/mean_length": 198.75, "completions/min_length": 195.0, "completions/max_length": 210.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 198.75, "completions/min_terminated_length": 195.0, "completions/max_terminated_length": 210.0, "rewards/meter/mean": 0.9345196485519409, "rewards/meter/std": 0.17161889374256134, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9137976765632629, "rewards/total_composite/std": 0.1733204871416092, "reward": 0.9137976765632629, "reward_std": 0.1733204573392868, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009508727118372917, "sampling/sampling_logp_difference/max": 1.1895999908447266, "sampling/importance_sampling_ratio/min": 0.30434298515319824, "sampling/importance_sampling_ratio/mean": 1.0017095804214478, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05790994456037879, "clip_ratio/low_mean": 0.004212898784317076, "clip_ratio/low_min": 0.004212898784317076, "clip_ratio/high_mean": 0.0064004448358900845, "clip_ratio/high_max": 0.0064004448358900845, "clip_ratio/region_mean": 0.01061334362020716, "reward_total_mean": 0.9137976765632629, "reward_meter_mean": 0.9345196485519409, "reward_meter_std": 0.17161889374256134, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9137976765632629, "reward_total_composite_std": 0.1733204871416092, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 616.0} {"timestamp_utc": "2026-04-11T20:41:11Z", "mode": "train", "global_step": 617, "epoch": 0.023826073524868707, "loss": 0.1404, "grad_norm": 3.8700666427612305, "learning_rate": 8.133333333333334e-06, "num_tokens": 1337009.0, "completions/mean_length": 189.375, "completions/min_length": 137.0, "completions/max_length": 456.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 189.375, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 456.0, "rewards/meter/mean": 0.12910132110118866, "rewards/meter/std": 0.23855595290660858, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.11995942890644073, "rewards/total_composite/std": 0.22783412039279938, "reward": 0.11995942890644073, "reward_std": 0.22783413529396057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08845033496618271, "sampling/sampling_logp_difference/max": 4.667322158813477, "sampling/importance_sampling_ratio/min": 0.009397400543093681, "sampling/importance_sampling_ratio/mean": 1.0223716497421265, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9195150500163436, "clip_ratio/low_mean": 0.026029597967863083, "clip_ratio/low_min": 0.026029597967863083, "clip_ratio/high_mean": 0.013311688497196883, "clip_ratio/high_max": 0.013311688497196883, "clip_ratio/region_mean": 0.039341286465059966, "reward_total_mean": 0.11995942890644073, "reward_meter_mean": 0.12910132110118866, "reward_meter_std": 0.23855595290660858, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.11995942890644073, "reward_total_composite_std": 0.22783412039279938, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 617.0} {"timestamp_utc": "2026-04-11T20:41:21Z", "mode": "train", "global_step": 618, "epoch": 0.02386468952734013, "loss": -0.0697, "grad_norm": 0.84645676612854, "learning_rate": 8.130303030303031e-06, "num_tokens": 1338388.0, "completions/mean_length": 339.375, "completions/min_length": 48.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 51.66666793823242, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.5469461679458618, "rewards/meter/std": 0.45063626766204834, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.37216296792030334, "rewards/total_composite/std": 0.5076146125793457, "reward": 0.37216296792030334, "reward_std": 0.5076145529747009, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07076752930879593, "sampling/sampling_logp_difference/max": 1.4148378372192383, "sampling/importance_sampling_ratio/min": 0.24296501278877258, "sampling/importance_sampling_ratio/mean": 1.009710669517517, "sampling/importance_sampling_ratio/max": 1.5025615692138672, "entropy": 0.19933610782027245, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.031583026982843876, "clip_ratio/high_max": 0.031583026982843876, "clip_ratio/region_mean": 0.031583026982843876, "reward_total_mean": 0.37216296792030334, "reward_meter_mean": 0.5469461679458618, "reward_meter_std": 0.45063626766204834, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.37216296792030334, "reward_total_composite_std": 0.5076146125793457, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 618.0} {"timestamp_utc": "2026-04-11T20:41:26Z", "mode": "train", "global_step": 619, "epoch": 0.023903305529811555, "loss": -0.0004, "grad_norm": 4.1318182945251465, "learning_rate": 8.127272727272728e-06, "num_tokens": 1340369.0, "completions/mean_length": 92.625, "completions/min_length": 90.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.625, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.7933108806610107, "rewards/meter/std": 0.2967395782470703, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7933108806610107, "rewards/total_composite/std": 0.2967395782470703, "reward": 0.7933108806610107, "reward_std": 0.2967395484447479, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026417352259159088, "sampling/sampling_logp_difference/max": 0.9867763519287109, "sampling/importance_sampling_ratio/min": 0.37277644872665405, "sampling/importance_sampling_ratio/mean": 1.001517415046692, "sampling/importance_sampling_ratio/max": 1.3991177082061768, "entropy": 0.1813750467263162, "clip_ratio/low_mean": 0.014778794953599572, "clip_ratio/low_min": 0.014778794953599572, "clip_ratio/high_mean": 0.0026461693923920393, "clip_ratio/high_max": 0.0026461693923920393, "clip_ratio/region_mean": 0.01742496434599161, "reward_total_mean": 0.7933108806610107, "reward_meter_mean": 0.7933108806610107, "reward_meter_std": 0.2967395782470703, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7933108806610107, "reward_total_composite_std": 0.2967395782470703, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 619.0} {"timestamp_utc": "2026-04-11T20:41:30Z", "mode": "train", "global_step": 620, "epoch": 0.02394192153228298, "loss": 0.0358, "grad_norm": 11.696823120117188, "learning_rate": 8.124242424242424e-06, "num_tokens": 1341808.0, "completions/mean_length": 33.875, "completions/min_length": 33.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.978384256362915, "rewards/meter/std": 0.029845381155610085, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.978384256362915, "rewards/total_composite/std": 0.029845381155610085, "reward": 0.978384256362915, "reward_std": 0.029845381155610085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11957071721553802, "sampling/sampling_logp_difference/max": 2.2146639823913574, "sampling/importance_sampling_ratio/min": 0.1091901957988739, "sampling/importance_sampling_ratio/mean": 1.007791519165039, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5899157114326954, "clip_ratio/low_mean": 0.011140820104628801, "clip_ratio/low_min": 0.011140820104628801, "clip_ratio/high_mean": 0.0851530022919178, "clip_ratio/high_max": 0.0851530022919178, "clip_ratio/region_mean": 0.0962938223965466, "reward_total_mean": 0.978384256362915, "reward_meter_mean": 0.978384256362915, "reward_meter_std": 0.029845381155610085, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.978384256362915, "reward_total_composite_std": 0.029845381155610085, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 620.0} {"timestamp_utc": "2026-04-11T20:41:35Z", "mode": "train", "global_step": 621, "epoch": 0.023980537534754404, "loss": 0.034, "grad_norm": 4.917610168457031, "learning_rate": 8.121212121212121e-06, "num_tokens": 1343571.0, "completions/mean_length": 72.375, "completions/min_length": 66.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9518520832061768, "rewards/meter/std": 0.032847389578819275, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9518520832061768, "rewards/total_composite/std": 0.032847389578819275, "reward": 0.9518520832061768, "reward_std": 0.032847389578819275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04176875576376915, "sampling/sampling_logp_difference/max": 2.8032267093658447, "sampling/importance_sampling_ratio/min": 0.06061416491866112, "sampling/importance_sampling_ratio/mean": 1.0048478841781616, "sampling/importance_sampling_ratio/max": 1.9684488773345947, "entropy": 0.2109714262187481, "clip_ratio/low_mean": 0.011628984240815043, "clip_ratio/low_min": 0.011628984240815043, "clip_ratio/high_mean": 0.017727577593177557, "clip_ratio/high_max": 0.017727577593177557, "clip_ratio/region_mean": 0.0293565618339926, "reward_total_mean": 0.9518520832061768, "reward_meter_mean": 0.9518520832061768, "reward_meter_std": 0.032847389578819275, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9518520832061768, "reward_total_composite_std": 0.032847389578819275, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 621.0} {"timestamp_utc": "2026-04-11T20:41:45Z", "mode": "train", "global_step": 622, "epoch": 0.024019153537225828, "loss": -0.1456, "grad_norm": 1.2342267036437988, "learning_rate": 8.118181818181819e-06, "num_tokens": 1345212.0, "completions/mean_length": 113.125, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 56.142860412597656, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8251349329948425, "rewards/meter/std": 0.3412078320980072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8246999979019165, "rewards/total_composite/std": 0.342404842376709, "reward": 0.8246999979019165, "reward_std": 0.342404842376709, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07290157675743103, "sampling/sampling_logp_difference/max": 1.4786067008972168, "sampling/importance_sampling_ratio/min": 0.22795508801937103, "sampling/importance_sampling_ratio/mean": 1.0137029886245728, "sampling/importance_sampling_ratio/max": 1.750449538230896, "entropy": 0.4394574910402298, "clip_ratio/low_mean": 0.0062500000931322575, "clip_ratio/low_min": 0.0062500000931322575, "clip_ratio/high_mean": 0.03944407729431987, "clip_ratio/high_max": 0.03944407729431987, "clip_ratio/region_mean": 0.045694077387452126, "reward_total_mean": 0.8246999979019165, "reward_meter_mean": 0.8251349329948425, "reward_meter_std": 0.3412078320980072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8246999979019165, "reward_total_composite_std": 0.342404842376709, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 622.0} {"timestamp_utc": "2026-04-11T20:41:55Z", "mode": "train", "global_step": 623, "epoch": 0.024057769539697252, "loss": -0.1006, "grad_norm": 1.7279670238494873, "learning_rate": 8.115151515151516e-06, "num_tokens": 1346949.0, "completions/mean_length": 125.125, "completions/min_length": 63.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 69.85714721679688, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.7928895950317383, "rewards/meter/std": 0.3771921992301941, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7471871376037598, "rewards/total_composite/std": 0.4512399435043335, "reward": 0.7471871376037598, "reward_std": 0.4512399137020111, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.050993990153074265, "sampling/sampling_logp_difference/max": 1.0117974281311035, "sampling/importance_sampling_ratio/min": 0.3635649085044861, "sampling/importance_sampling_ratio/mean": 1.0057697296142578, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31833127327263355, "clip_ratio/low_mean": 0.001623376621864736, "clip_ratio/low_min": 0.001623376621864736, "clip_ratio/high_mean": 0.025720802485011518, "clip_ratio/high_max": 0.025720802485011518, "clip_ratio/region_mean": 0.027344179106876254, "reward_total_mean": 0.7471871376037598, "reward_meter_mean": 0.7928895950317383, "reward_meter_std": 0.3771921992301941, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7471871376037598, "reward_total_composite_std": 0.4512399435043335, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 623.0} {"timestamp_utc": "2026-04-11T20:42:01Z", "mode": "train", "global_step": 624, "epoch": 0.024096385542168676, "loss": -0.0178, "grad_norm": 2.252504825592041, "learning_rate": 8.112121212121213e-06, "num_tokens": 1350067.0, "completions/mean_length": 182.75, "completions/min_length": 181.0, "completions/max_length": 192.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 182.75, "completions/min_terminated_length": 181.0, "completions/max_terminated_length": 192.0, "rewards/meter/mean": 0.9957711100578308, "rewards/meter/std": 0.0014193379320204258, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9957711100578308, "rewards/total_composite/std": 0.0014193379320204258, "reward": 0.9957711100578308, "reward_std": 0.0014193379320204258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005115547217428684, "sampling/sampling_logp_difference/max": 0.6170334815979004, "sampling/importance_sampling_ratio/min": 0.5395426154136658, "sampling/importance_sampling_ratio/mean": 1.0003677606582642, "sampling/importance_sampling_ratio/max": 1.406175971031189, "entropy": 0.04385493486188352, "clip_ratio/low_mean": 0.0013736264081671834, "clip_ratio/low_min": 0.0013736264081671834, "clip_ratio/high_mean": 0.0006510416860692203, "clip_ratio/high_max": 0.0006510416860692203, "clip_ratio/region_mean": 0.0020246680942364037, "reward_total_mean": 0.9957711100578308, "reward_meter_mean": 0.9957711100578308, "reward_meter_std": 0.0014193379320204258, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9957711100578308, "reward_total_composite_std": 0.0014193379320204258, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 624.0} {"timestamp_utc": "2026-04-11T20:42:11Z", "mode": "train", "global_step": 625, "epoch": 0.0241350015446401, "loss": -0.1074, "grad_norm": 2.2000861167907715, "learning_rate": 8.10909090909091e-06, "num_tokens": 1353587.0, "completions/mean_length": 297.0, "completions/min_length": 247.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 266.2857360839844, "completions/min_terminated_length": 247.0, "completions/max_terminated_length": 288.0, "rewards/meter/mean": 0.5770298838615417, "rewards/meter/std": 0.47208163142204285, "rewards/count_adherence/mean": 0.9027777910232544, "rewards/count_adherence/std": 0.03928370773792267, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.4286315441131592, "rewards/total_composite/std": 0.44896507263183594, "reward": 0.4286315441131592, "reward_std": 0.44896507263183594, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021907327696681023, "sampling/sampling_logp_difference/max": 1.469860553741455, "sampling/importance_sampling_ratio/min": 0.22995755076408386, "sampling/importance_sampling_ratio/mean": 1.0011159181594849, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09663973702117801, "clip_ratio/low_mean": 0.005678939749486744, "clip_ratio/low_min": 0.005678939749486744, "clip_ratio/high_mean": 0.00999945483636111, "clip_ratio/high_max": 0.00999945483636111, "clip_ratio/region_mean": 0.015678394585847855, "reward_total_mean": 0.4286315441131592, "reward_meter_mean": 0.5770298838615417, "reward_meter_std": 0.47208163142204285, "reward_count_adherence_mean": 0.9027777910232544, "reward_count_adherence_std": 0.03928370773792267, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.4286315441131592, "reward_total_composite_std": 0.44896507263183594, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 625.0} {"timestamp_utc": "2026-04-11T20:42:18Z", "mode": "train", "global_step": 626, "epoch": 0.024173617547111524, "loss": -0.0186, "grad_norm": 1.3433260917663574, "learning_rate": 8.106060606060606e-06, "num_tokens": 1357445.0, "completions/mean_length": 258.25, "completions/min_length": 239.0, "completions/max_length": 272.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 258.25, "completions/min_terminated_length": 239.0, "completions/max_terminated_length": 272.0, "rewards/meter/mean": 0.9910991787910461, "rewards/meter/std": 0.006634681485593319, "rewards/count_adherence/mean": 0.921875, "rewards/count_adherence/std": 0.06469365209341049, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9136239886283875, "rewards/total_composite/std": 0.06364713609218597, "reward": 0.9136239886283875, "reward_std": 0.06364713609218597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01168814580887556, "sampling/sampling_logp_difference/max": 1.3464345932006836, "sampling/importance_sampling_ratio/min": 0.2601662278175354, "sampling/importance_sampling_ratio/mean": 1.0006234645843506, "sampling/importance_sampling_ratio/max": 1.596511960029602, "entropy": 0.08377446280792356, "clip_ratio/low_mean": 0.005793766817077994, "clip_ratio/low_min": 0.005793766817077994, "clip_ratio/high_mean": 0.002810997946653515, "clip_ratio/high_max": 0.002810997946653515, "clip_ratio/region_mean": 0.00860476476373151, "reward_total_mean": 0.9136239886283875, "reward_meter_mean": 0.9910991787910461, "reward_meter_std": 0.006634681485593319, "reward_count_adherence_mean": 0.921875, "reward_count_adherence_std": 0.06469365209341049, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9136239886283875, "reward_total_composite_std": 0.06364713609218597, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 626.0} {"timestamp_utc": "2026-04-11T20:42:28Z", "mode": "train", "global_step": 627, "epoch": 0.02421223354958295, "loss": -0.1003, "grad_norm": 1.754539966583252, "learning_rate": 8.103030303030303e-06, "num_tokens": 1359055.0, "completions/mean_length": 175.25, "completions/min_length": 51.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.7318032383918762, "rewards/meter/std": 0.3413369357585907, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.5797286033630371, "rewards/total_composite/std": 0.4377219080924988, "reward": 0.5797286033630371, "reward_std": 0.4377219080924988, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11965037882328033, "sampling/sampling_logp_difference/max": 1.3655691146850586, "sampling/importance_sampling_ratio/min": 0.25523537397384644, "sampling/importance_sampling_ratio/mean": 1.012277603149414, "sampling/importance_sampling_ratio/max": 1.7900029420852661, "entropy": 0.9980961419641972, "clip_ratio/low_mean": 0.010802469216287136, "clip_ratio/low_min": 0.010802469216287136, "clip_ratio/high_mean": 0.04422784689813852, "clip_ratio/high_max": 0.04422784689813852, "clip_ratio/region_mean": 0.05503031611442566, "reward_total_mean": 0.5797286033630371, "reward_meter_mean": 0.7318032383918762, "reward_meter_std": 0.3413369357585907, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.25877460837364197, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.5797286033630371, "reward_total_composite_std": 0.4377219080924988, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 627.0} {"timestamp_utc": "2026-04-11T20:42:33Z", "mode": "train", "global_step": 628, "epoch": 0.024250849552054372, "loss": 0.0687, "grad_norm": 5.110913276672363, "learning_rate": 8.1e-06, "num_tokens": 1361096.0, "completions/mean_length": 92.125, "completions/min_length": 89.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.125, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.970349907875061, "rewards/meter/std": 0.06832630932331085, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.970349907875061, "rewards/total_composite/std": 0.06832630932331085, "reward": 0.970349907875061, "reward_std": 0.06832629442214966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02751935087144375, "sampling/sampling_logp_difference/max": 1.1623228788375854, "sampling/importance_sampling_ratio/min": 0.31275883316993713, "sampling/importance_sampling_ratio/mean": 0.9971374273300171, "sampling/importance_sampling_ratio/max": 1.841081142425537, "entropy": 0.19556331355124712, "clip_ratio/low_mean": 0.0069444444961845875, "clip_ratio/low_min": 0.0069444444961845875, "clip_ratio/high_mean": 0.020865230937488377, "clip_ratio/high_max": 0.020865230937488377, "clip_ratio/region_mean": 0.027809675433672965, "reward_total_mean": 0.970349907875061, "reward_meter_mean": 0.970349907875061, "reward_meter_std": 0.06832630932331085, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.970349907875061, "reward_total_composite_std": 0.06832630932331085, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 628.0} {"timestamp_utc": "2026-04-11T20:42:43Z", "mode": "train", "global_step": 629, "epoch": 0.024289465554525796, "loss": -0.1464, "grad_norm": 0.993527352809906, "learning_rate": 8.096969696969698e-06, "num_tokens": 1362727.0, "completions/mean_length": 182.875, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 73.16667175292969, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9057246446609497, "rewards/meter/std": 0.21483053267002106, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.662638783454895, "rewards/total_composite/std": 0.46009400486946106, "reward": 0.662638783454895, "reward_std": 0.46009397506713867, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0446912907063961, "sampling/sampling_logp_difference/max": 0.8761682510375977, "sampling/importance_sampling_ratio/min": 0.46451571583747864, "sampling/importance_sampling_ratio/mean": 1.0186997652053833, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30349646881222725, "clip_ratio/low_mean": 0.010869565419852734, "clip_ratio/low_min": 0.010869565419852734, "clip_ratio/high_mean": 0.016275564790703356, "clip_ratio/high_max": 0.016275564790703356, "clip_ratio/region_mean": 0.02714513021055609, "reward_total_mean": 0.662638783454895, "reward_meter_mean": 0.9057246446609497, "reward_meter_std": 0.21483053267002106, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.662638783454895, "reward_total_composite_std": 0.46009400486946106, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 629.0} {"timestamp_utc": "2026-04-11T20:42:48Z", "mode": "train", "global_step": 630, "epoch": 0.02432808155699722, "loss": 0.0194, "grad_norm": 8.351807594299316, "learning_rate": 8.093939393939395e-06, "num_tokens": 1364463.0, "completions/mean_length": 62.0, "completions/min_length": 60.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9843227863311768, "rewards/meter/std": 0.027253830805420876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9843227863311768, "rewards/total_composite/std": 0.027253830805420876, "reward": 0.9843227863311768, "reward_std": 0.027253834530711174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05221335589885712, "sampling/sampling_logp_difference/max": 1.4105112552642822, "sampling/importance_sampling_ratio/min": 0.24401849508285522, "sampling/importance_sampling_ratio/mean": 1.0087597370147705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25993255618959665, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.04845884535461664, "clip_ratio/high_max": 0.04845884535461664, "clip_ratio/region_mean": 0.05236509535461664, "reward_total_mean": 0.9843227863311768, "reward_meter_mean": 0.9843227863311768, "reward_meter_std": 0.027253830805420876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9843227863311768, "reward_total_composite_std": 0.027253830805420876, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 630.0} {"timestamp_utc": "2026-04-11T20:42:58Z", "mode": "train", "global_step": 631, "epoch": 0.024366697559468645, "loss": -0.1755, "grad_norm": 0.8814429640769958, "learning_rate": 8.090909090909092e-06, "num_tokens": 1366217.0, "completions/mean_length": 127.25, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 72.28572082519531, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8845906257629395, "rewards/meter/std": 0.2908637821674347, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8639343976974487, "rewards/total_composite/std": 0.3492541015148163, "reward": 0.8639343976974487, "reward_std": 0.3492541015148163, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028667865321040154, "sampling/sampling_logp_difference/max": 0.8945565223693848, "sampling/importance_sampling_ratio/min": 0.4087888300418854, "sampling/importance_sampling_ratio/mean": 1.006989598274231, "sampling/importance_sampling_ratio/max": 1.5585085153579712, "entropy": 0.1889364793896675, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.011983279720880091, "clip_ratio/high_max": 0.011983279720880091, "clip_ratio/region_mean": 0.011983279720880091, "reward_total_mean": 0.8639343976974487, "reward_meter_mean": 0.8845906257629395, "reward_meter_std": 0.2908637821674347, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8639343976974487, "reward_total_composite_std": 0.3492541015148163, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 631.0} {"timestamp_utc": "2026-04-11T20:43:03Z", "mode": "train", "global_step": 632, "epoch": 0.02440531356194007, "loss": 0.0991, "grad_norm": 3.513773202896118, "learning_rate": 8.08787878787879e-06, "num_tokens": 1368537.0, "completions/mean_length": 111.0, "completions/min_length": 99.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.0, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.9886435270309448, "rewards/meter/std": 0.0022036279551684856, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9886435270309448, "rewards/total_composite/std": 0.0022036279551684856, "reward": 0.9886435270309448, "reward_std": 0.0022036198060959578, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01080810371786356, "sampling/sampling_logp_difference/max": 0.9212517142295837, "sampling/importance_sampling_ratio/min": 0.39802050590515137, "sampling/importance_sampling_ratio/mean": 1.0014369487762451, "sampling/importance_sampling_ratio/max": 1.52896249294281, "entropy": 0.07266335096210241, "clip_ratio/low_mean": 0.0009259259095415473, "clip_ratio/low_min": 0.0009259259095415473, "clip_ratio/high_mean": 0.007443342707119882, "clip_ratio/high_max": 0.007443342707119882, "clip_ratio/region_mean": 0.00836926861666143, "reward_total_mean": 0.9886435270309448, "reward_meter_mean": 0.9886435270309448, "reward_meter_std": 0.0022036279551684856, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9886435270309448, "reward_total_composite_std": 0.0022036279551684856, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 632.0} {"timestamp_utc": "2026-04-11T20:43:09Z", "mode": "train", "global_step": 633, "epoch": 0.024443929564411493, "loss": 0.0485, "grad_norm": 3.3415863513946533, "learning_rate": 8.084848484848485e-06, "num_tokens": 1370640.0, "completions/mean_length": 84.875, "completions/min_length": 81.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.875, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.8842945098876953, "rewards/meter/std": 0.31306833028793335, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8842945098876953, "rewards/total_composite/std": 0.31306833028793335, "reward": 0.8842945098876953, "reward_std": 0.31306830048561096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02020621858537197, "sampling/sampling_logp_difference/max": 1.4568853378295898, "sampling/importance_sampling_ratio/min": 0.2329607456922531, "sampling/importance_sampling_ratio/mean": 1.005111813545227, "sampling/importance_sampling_ratio/max": 1.9100323915481567, "entropy": 0.17416242323815823, "clip_ratio/low_mean": 0.0013020833721384406, "clip_ratio/low_min": 0.0013020833721384406, "clip_ratio/high_mean": 0.022631968837231398, "clip_ratio/high_max": 0.022631968837231398, "clip_ratio/region_mean": 0.023934052209369838, "reward_total_mean": 0.8842945098876953, "reward_meter_mean": 0.8842945098876953, "reward_meter_std": 0.31306833028793335, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8842945098876953, "reward_total_composite_std": 0.31306833028793335, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 633.0} {"timestamp_utc": "2026-04-11T20:43:15Z", "mode": "train", "global_step": 634, "epoch": 0.024482545566882917, "loss": 0.0184, "grad_norm": 2.963343858718872, "learning_rate": 8.081818181818182e-06, "num_tokens": 1372994.0, "completions/mean_length": 126.25, "completions/min_length": 115.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.25, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.9853245615959167, "rewards/meter/std": 0.021229863166809082, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9853245615959167, "rewards/total_composite/std": 0.021229863166809082, "reward": 0.9853245615959167, "reward_std": 0.02122986875474453, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027480771765112877, "sampling/sampling_logp_difference/max": 1.187922477722168, "sampling/importance_sampling_ratio/min": 0.3048539459705353, "sampling/importance_sampling_ratio/mean": 1.0044063329696655, "sampling/importance_sampling_ratio/max": 1.6705741882324219, "entropy": 0.20136078912764788, "clip_ratio/low_mean": 0.012231739703565836, "clip_ratio/low_min": 0.012231739703565836, "clip_ratio/high_mean": 0.018129791948013008, "clip_ratio/high_max": 0.018129791948013008, "clip_ratio/region_mean": 0.030361531651578844, "reward_total_mean": 0.9853245615959167, "reward_meter_mean": 0.9853245615959167, "reward_meter_std": 0.021229863166809082, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9853245615959167, "reward_total_composite_std": 0.021229863166809082, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 634.0} {"timestamp_utc": "2026-04-11T20:43:20Z", "mode": "train", "global_step": 635, "epoch": 0.02452116156935434, "loss": -0.0067, "grad_norm": 6.4878387451171875, "learning_rate": 8.07878787878788e-06, "num_tokens": 1374785.0, "completions/mean_length": 61.875, "completions/min_length": 60.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9954343438148499, "rewards/meter/std": 0.0015961348544806242, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9954343438148499, "rewards/total_composite/std": 0.0015961348544806242, "reward": 0.9954343438148499, "reward_std": 0.0015961244935169816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0258852057158947, "sampling/sampling_logp_difference/max": 1.8346977233886719, "sampling/importance_sampling_ratio/min": 0.15966175496578217, "sampling/importance_sampling_ratio/mean": 0.9988678693771362, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11720475321635604, "clip_ratio/low_mean": 0.007863856852054596, "clip_ratio/low_min": 0.007863856852054596, "clip_ratio/high_mean": 0.012339744134806097, "clip_ratio/high_max": 0.012339744134806097, "clip_ratio/region_mean": 0.020203600986860693, "reward_total_mean": 0.9954343438148499, "reward_meter_mean": 0.9954343438148499, "reward_meter_std": 0.0015961348544806242, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9954343438148499, "reward_total_composite_std": 0.0015961348544806242, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 635.0} {"timestamp_utc": "2026-04-11T20:43:26Z", "mode": "train", "global_step": 636, "epoch": 0.024559777571825765, "loss": -0.0179, "grad_norm": 1.9705545902252197, "learning_rate": 8.075757575757577e-06, "num_tokens": 1377307.0, "completions/mean_length": 122.25, "completions/min_length": 121.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.25, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9956600069999695, "rewards/meter/std": 0.001489842776209116, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9956600069999695, "rewards/total_composite/std": 0.001489842776209116, "reward": 0.9956600069999695, "reward_std": 0.001489846152253449, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0057179187424480915, "sampling/sampling_logp_difference/max": 2.2857398986816406, "sampling/importance_sampling_ratio/min": 0.10169878602027893, "sampling/importance_sampling_ratio/mean": 1.0008108615875244, "sampling/importance_sampling_ratio/max": 1.2143502235412598, "entropy": 0.03814642573706806, "clip_ratio/low_mean": 0.0020576479146257043, "clip_ratio/low_min": 0.0020576479146257043, "clip_ratio/high_mean": 0.002884615445509553, "clip_ratio/high_max": 0.002884615445509553, "clip_ratio/region_mean": 0.004942263360135257, "reward_total_mean": 0.9956600069999695, "reward_meter_mean": 0.9956600069999695, "reward_meter_std": 0.001489842776209116, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9956600069999695, "reward_total_composite_std": 0.001489842776209116, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 636.0} {"timestamp_utc": "2026-04-11T20:43:32Z", "mode": "train", "global_step": 637, "epoch": 0.02459839357429719, "loss": 0.0233, "grad_norm": 2.965711832046509, "learning_rate": 8.072727272727274e-06, "num_tokens": 1379671.0, "completions/mean_length": 123.5, "completions/min_length": 112.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.5, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9958841800689697, "rewards/meter/std": 0.0024583181366324425, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9958841800689697, "rewards/total_composite/std": 0.0024583181366324425, "reward": 0.9958841800689697, "reward_std": 0.002458317205309868, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027591824531555176, "sampling/sampling_logp_difference/max": 0.9434237480163574, "sampling/importance_sampling_ratio/min": 0.38929271697998047, "sampling/importance_sampling_ratio/mean": 1.0019030570983887, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18402807414531708, "clip_ratio/low_mean": 0.003179112682119012, "clip_ratio/low_min": 0.003179112682119012, "clip_ratio/high_mean": 0.010326079092919827, "clip_ratio/high_max": 0.010326079092919827, "clip_ratio/region_mean": 0.013505191775038838, "reward_total_mean": 0.9958841800689697, "reward_meter_mean": 0.9958841800689697, "reward_meter_std": 0.0024583181366324425, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9958841800689697, "reward_total_composite_std": 0.0024583181366324425, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 637.0} {"timestamp_utc": "2026-04-11T20:43:36Z", "mode": "train", "global_step": 638, "epoch": 0.024637009576768613, "loss": -0.0238, "grad_norm": 11.34463882446289, "learning_rate": 8.069696969696971e-06, "num_tokens": 1381546.0, "completions/mean_length": 64.375, "completions/min_length": 56.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9862507581710815, "rewards/meter/std": 0.01588655635714531, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9862507581710815, "rewards/total_composite/std": 0.01588655635714531, "reward": 0.9862507581710815, "reward_std": 0.015886547043919563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08512242883443832, "sampling/sampling_logp_difference/max": 2.88051176071167, "sampling/importance_sampling_ratio/min": 0.05610604211688042, "sampling/importance_sampling_ratio/mean": 1.005232810974121, "sampling/importance_sampling_ratio/max": 1.9803534746170044, "entropy": 0.5689874831587076, "clip_ratio/low_mean": 0.023809524718672037, "clip_ratio/low_min": 0.023809524718672037, "clip_ratio/high_mean": 0.051263275323435664, "clip_ratio/high_max": 0.051263275323435664, "clip_ratio/region_mean": 0.0750728000421077, "reward_total_mean": 0.9862507581710815, "reward_meter_mean": 0.9862507581710815, "reward_meter_std": 0.01588655635714531, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9862507581710815, "reward_total_composite_std": 0.01588655635714531, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 638.0} {"timestamp_utc": "2026-04-11T20:43:41Z", "mode": "train", "global_step": 639, "epoch": 0.024675625579240038, "loss": -0.0012, "grad_norm": 5.633913993835449, "learning_rate": 8.066666666666667e-06, "num_tokens": 1383384.0, "completions/mean_length": 74.75, "completions/min_length": 70.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.75, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.8388459086418152, "rewards/meter/std": 0.19675958156585693, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8388459086418152, "rewards/total_composite/std": 0.19675958156585693, "reward": 0.8388459086418152, "reward_std": 0.19675958156585693, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05180966109037399, "sampling/sampling_logp_difference/max": 1.5161762237548828, "sampling/importance_sampling_ratio/min": 0.21954980492591858, "sampling/importance_sampling_ratio/mean": 0.998264491558075, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3610265199095011, "clip_ratio/low_mean": 0.015207641525194049, "clip_ratio/low_min": 0.015207641525194049, "clip_ratio/high_mean": 0.026776819955557585, "clip_ratio/high_max": 0.026776819955557585, "clip_ratio/region_mean": 0.041984461480751634, "reward_total_mean": 0.8388459086418152, "reward_meter_mean": 0.8388459086418152, "reward_meter_std": 0.19675958156585693, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8388459086418152, "reward_total_composite_std": 0.19675958156585693, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 639.0} {"timestamp_utc": "2026-04-11T20:43:48Z", "mode": "train", "global_step": 640, "epoch": 0.02471424158171146, "loss": 0.0423, "grad_norm": 0.2766904830932617, "learning_rate": 8.063636363636364e-06, "num_tokens": 1386090.0, "completions/mean_length": 162.25, "completions/min_length": 151.0, "completions/max_length": 181.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 162.25, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.9950998425483704, "rewards/meter/std": 0.0002485640870872885, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.970228910446167, "rewards/total_composite/std": 0.0704522356390953, "reward": 0.970228910446167, "reward_std": 0.0704522356390953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007383763324469328, "sampling/sampling_logp_difference/max": 2.5497994422912598, "sampling/importance_sampling_ratio/min": 0.07809732854366302, "sampling/importance_sampling_ratio/mean": 1.0019837617874146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.01887570507824421, "clip_ratio/low_mean": 0.002071823226287961, "clip_ratio/low_min": 0.002071823226287961, "clip_ratio/high_mean": 0.0054954917868599296, "clip_ratio/high_max": 0.0054954917868599296, "clip_ratio/region_mean": 0.007567315013147891, "reward_total_mean": 0.970228910446167, "reward_meter_mean": 0.9950998425483704, "reward_meter_std": 0.0002485640870872885, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.970228910446167, "reward_total_composite_std": 0.0704522356390953, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 640.0} {"timestamp_utc": "2026-04-11T20:43:53Z", "mode": "train", "global_step": 641, "epoch": 0.024752857584182886, "loss": -0.0001, "grad_norm": 3.8018136024475098, "learning_rate": 8.060606060606061e-06, "num_tokens": 1387880.0, "completions/mean_length": 76.75, "completions/min_length": 72.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.75, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9926354885101318, "rewards/meter/std": 0.003230726346373558, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9926354885101318, "rewards/total_composite/std": 0.003230726346373558, "reward": 0.9926354885101318, "reward_std": 0.0032307212240993977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03708671033382416, "sampling/sampling_logp_difference/max": 0.9711356163024902, "sampling/importance_sampling_ratio/min": 0.37865278124809265, "sampling/importance_sampling_ratio/mean": 1.0071512460708618, "sampling/importance_sampling_ratio/max": 1.8937081098556519, "entropy": 0.1969589926302433, "clip_ratio/low_mean": 0.012743587838485837, "clip_ratio/low_min": 0.012743587838485837, "clip_ratio/high_mean": 0.025679388898424804, "clip_ratio/high_max": 0.025679388898424804, "clip_ratio/region_mean": 0.03842297673691064, "reward_total_mean": 0.9926354885101318, "reward_meter_mean": 0.9926354885101318, "reward_meter_std": 0.003230726346373558, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9926354885101318, "reward_total_composite_std": 0.003230726346373558, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 641.0} {"timestamp_utc": "2026-04-11T20:43:57Z", "mode": "train", "global_step": 642, "epoch": 0.02479147358665431, "loss": 0.0512, "grad_norm": 10.998772621154785, "learning_rate": 8.057575757575759e-06, "num_tokens": 1389306.0, "completions/mean_length": 40.25, "completions/min_length": 34.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9872698783874512, "rewards/meter/std": 0.01214898843318224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9872698783874512, "rewards/total_composite/std": 0.01214898843318224, "reward": 0.9872698783874512, "reward_std": 0.012148984707891941, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0713094025850296, "sampling/sampling_logp_difference/max": 1.2929267883300781, "sampling/importance_sampling_ratio/min": 0.27446630597114563, "sampling/importance_sampling_ratio/mean": 1.0039377212524414, "sampling/importance_sampling_ratio/max": 1.8077318668365479, "entropy": 0.34129922464489937, "clip_ratio/low_mean": 0.012456294149160385, "clip_ratio/low_min": 0.012456294149160385, "clip_ratio/high_mean": 0.053032459458336234, "clip_ratio/high_max": 0.053032459458336234, "clip_ratio/region_mean": 0.06548875360749662, "reward_total_mean": 0.9872698783874512, "reward_meter_mean": 0.9872698783874512, "reward_meter_std": 0.01214898843318224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9872698783874512, "reward_total_composite_std": 0.01214898843318224, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 642.0} {"timestamp_utc": "2026-04-11T20:44:05Z", "mode": "train", "global_step": 643, "epoch": 0.024830089589125734, "loss": 0.11, "grad_norm": 3.1086318492889404, "learning_rate": 8.054545454545454e-06, "num_tokens": 1392997.0, "completions/mean_length": 246.375, "completions/min_length": 192.0, "completions/max_length": 313.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 246.375, "completions/min_terminated_length": 192.0, "completions/max_terminated_length": 313.0, "rewards/meter/mean": 0.3895261585712433, "rewards/meter/std": 0.42622098326683044, "rewards/count_adherence/mean": 0.930555522441864, "rewards/count_adherence/std": 0.05750546231865883, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3722308278083801, "rewards/total_composite/std": 0.41810813546180725, "reward": 0.3722308278083801, "reward_std": 0.41810813546180725, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02562355063855648, "sampling/sampling_logp_difference/max": 3.0642731189727783, "sampling/importance_sampling_ratio/min": 0.046687766909599304, "sampling/importance_sampling_ratio/mean": 1.0075058937072754, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11560235964134336, "clip_ratio/low_mean": 0.011697308160364628, "clip_ratio/low_min": 0.011697308160364628, "clip_ratio/high_mean": 0.0059894436853937805, "clip_ratio/high_max": 0.0059894436853937805, "clip_ratio/region_mean": 0.01768675184575841, "reward_total_mean": 0.3722308278083801, "reward_meter_mean": 0.3895261585712433, "reward_meter_std": 0.42622098326683044, "reward_count_adherence_mean": 0.930555522441864, "reward_count_adherence_std": 0.05750546231865883, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3722308278083801, "reward_total_composite_std": 0.41810813546180725, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 643.0} {"timestamp_utc": "2026-04-11T20:44:15Z", "mode": "train", "global_step": 644, "epoch": 0.024868705591597158, "loss": -0.1547, "grad_norm": 0.7139359712600708, "learning_rate": 8.051515151515153e-06, "num_tokens": 1394733.0, "completions/mean_length": 114.0, "completions/min_length": 55.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.142860412597656, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8798133134841919, "rewards/meter/std": 0.321513295173645, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8692982196807861, "rewards/total_composite/std": 0.3512541353702545, "reward": 0.8692982196807861, "reward_std": 0.3512541353702545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0238660741597414, "sampling/sampling_logp_difference/max": 1.2294301986694336, "sampling/importance_sampling_ratio/min": 0.2924591600894928, "sampling/importance_sampling_ratio/mean": 0.9990324378013611, "sampling/importance_sampling_ratio/max": 1.4886727333068848, "entropy": 0.13272713590413332, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0217176906298846, "clip_ratio/high_max": 0.0217176906298846, "clip_ratio/region_mean": 0.0217176906298846, "reward_total_mean": 0.8692982196807861, "reward_meter_mean": 0.8798133134841919, "reward_meter_std": 0.321513295173645, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8692982196807861, "reward_total_composite_std": 0.3512541353702545, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 644.0} {"timestamp_utc": "2026-04-11T20:44:20Z", "mode": "train", "global_step": 645, "epoch": 0.024907321594068582, "loss": -0.0075, "grad_norm": 2.8207523822784424, "learning_rate": 8.048484848484849e-06, "num_tokens": 1397215.0, "completions/mean_length": 127.25, "completions/min_length": 119.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.25, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.5935784578323364, "rewards/meter/std": 0.49478766322135925, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5935784578323364, "rewards/total_composite/std": 0.49478766322135925, "reward": 0.5935784578323364, "reward_std": 0.49478766322135925, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0225541889667511, "sampling/sampling_logp_difference/max": 1.666172981262207, "sampling/importance_sampling_ratio/min": 0.18896886706352234, "sampling/importance_sampling_ratio/mean": 1.0039057731628418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07990281283855438, "clip_ratio/low_mean": 0.007154436782002449, "clip_ratio/low_min": 0.007154436782002449, "clip_ratio/high_mean": 0.008699847152456641, "clip_ratio/high_max": 0.008699847152456641, "clip_ratio/region_mean": 0.01585428393445909, "reward_total_mean": 0.5935784578323364, "reward_meter_mean": 0.5935784578323364, "reward_meter_std": 0.49478766322135925, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5935784578323364, "reward_total_composite_std": 0.49478766322135925, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 645.0} {"timestamp_utc": "2026-04-11T20:44:31Z", "mode": "train", "global_step": 646, "epoch": 0.024945937596540006, "loss": -0.0732, "grad_norm": 1.6291898488998413, "learning_rate": 8.045454545454546e-06, "num_tokens": 1399037.0, "completions/mean_length": 172.75, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 59.66666793823242, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.5550702810287476, "rewards/meter/std": 0.4581260085105896, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.49611878395080566, "rewards/total_composite/std": 0.49892622232437134, "reward": 0.49611878395080566, "reward_std": 0.49892619252204895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05649138242006302, "sampling/sampling_logp_difference/max": 1.5663788318634033, "sampling/importance_sampling_ratio/min": 0.22186465561389923, "sampling/importance_sampling_ratio/mean": 1.0126827955245972, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23534639924764633, "clip_ratio/low_mean": 0.015636918134987354, "clip_ratio/low_min": 0.015636918134987354, "clip_ratio/high_mean": 0.012931034667417407, "clip_ratio/high_max": 0.012931034667417407, "clip_ratio/region_mean": 0.02856795280240476, "reward_total_mean": 0.49611878395080566, "reward_meter_mean": 0.5550702810287476, "reward_meter_std": 0.4581260085105896, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.49611878395080566, "reward_total_composite_std": 0.49892622232437134, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 646.0} {"timestamp_utc": "2026-04-11T20:44:37Z", "mode": "train", "global_step": 647, "epoch": 0.02498455359901143, "loss": 0.0108, "grad_norm": 10.47767162322998, "learning_rate": 8.042424242424243e-06, "num_tokens": 1400768.0, "completions/mean_length": 66.375, "completions/min_length": 62.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9769783020019531, "rewards/meter/std": 0.051688529551029205, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9769783020019531, "rewards/total_composite/std": 0.051688529551029205, "reward": 0.9769783020019531, "reward_std": 0.05168852210044861, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.056386951357126236, "sampling/sampling_logp_difference/max": 1.5775794982910156, "sampling/importance_sampling_ratio/min": 0.20647425949573517, "sampling/importance_sampling_ratio/mean": 0.9968314170837402, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2588699162006378, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.05566767114214599, "clip_ratio/high_max": 0.05566767114214599, "clip_ratio/region_mean": 0.05753334274049848, "reward_total_mean": 0.9769783020019531, "reward_meter_mean": 0.9769783020019531, "reward_meter_std": 0.051688529551029205, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9769783020019531, "reward_total_composite_std": 0.051688529551029205, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 647.0} {"timestamp_utc": "2026-04-11T20:44:44Z", "mode": "train", "global_step": 648, "epoch": 0.025023169601482854, "loss": 0.0133, "grad_norm": 2.8141632080078125, "learning_rate": 8.03939393939394e-06, "num_tokens": 1404355.0, "completions/mean_length": 250.375, "completions/min_length": 222.0, "completions/max_length": 276.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 250.375, "completions/min_terminated_length": 222.0, "completions/max_terminated_length": 276.0, "rewards/meter/mean": 0.7232567071914673, "rewards/meter/std": 0.3504089117050171, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.07715168595314026, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6260231733322144, "rewards/total_composite/std": 0.3030511140823364, "reward": 0.6260231733322144, "reward_std": 0.3030511140823364, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026500631123781204, "sampling/sampling_logp_difference/max": 5.308352470397949, "sampling/importance_sampling_ratio/min": 0.004950075875967741, "sampling/importance_sampling_ratio/mean": 0.9957653880119324, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06133082904852927, "clip_ratio/low_mean": 0.005490340990945697, "clip_ratio/low_min": 0.005490340990945697, "clip_ratio/high_mean": 0.009631438297219574, "clip_ratio/high_max": 0.009631438297219574, "clip_ratio/region_mean": 0.015121779288165271, "reward_total_mean": 0.6260231733322144, "reward_meter_mean": 0.7232567071914673, "reward_meter_std": 0.3504089117050171, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.07715168595314026, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6260231733322144, "reward_total_composite_std": 0.3030511140823364, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 648.0} {"timestamp_utc": "2026-04-11T20:44:50Z", "mode": "train", "global_step": 649, "epoch": 0.02506178560395428, "loss": 0.0372, "grad_norm": 3.6201226711273193, "learning_rate": 8.036363636363636e-06, "num_tokens": 1406589.0, "completions/mean_length": 112.25, "completions/min_length": 102.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.25, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.981624960899353, "rewards/meter/std": 0.027504365891218185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.981624960899353, "rewards/total_composite/std": 0.027504365891218185, "reward": 0.981624960899353, "reward_std": 0.02750437520444393, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019951220601797104, "sampling/sampling_logp_difference/max": 1.1682255268096924, "sampling/importance_sampling_ratio/min": 0.3109181523323059, "sampling/importance_sampling_ratio/mean": 1.0045708417892456, "sampling/importance_sampling_ratio/max": 1.9342072010040283, "entropy": 0.11915135849267244, "clip_ratio/low_mean": 0.006225490127690136, "clip_ratio/low_min": 0.006225490127690136, "clip_ratio/high_mean": 0.011183355352841318, "clip_ratio/high_max": 0.011183355352841318, "clip_ratio/region_mean": 0.017408845480531454, "reward_total_mean": 0.981624960899353, "reward_meter_mean": 0.981624960899353, "reward_meter_std": 0.027504365891218185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.981624960899353, "reward_total_composite_std": 0.027504365891218185, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 649.0} {"timestamp_utc": "2026-04-11T20:44:55Z", "mode": "train", "global_step": 650, "epoch": 0.025100401606425703, "loss": 0.0674, "grad_norm": 12.612115859985352, "learning_rate": 8.033333333333335e-06, "num_tokens": 1408050.0, "completions/mean_length": 35.625, "completions/min_length": 30.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9578008651733398, "rewards/meter/std": 0.05439894273877144, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9578008651733398, "rewards/total_composite/std": 0.05439894273877144, "reward": 0.9578008651733398, "reward_std": 0.05439893528819084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11987494677305222, "sampling/sampling_logp_difference/max": 1.6802425384521484, "sampling/importance_sampling_ratio/min": 0.18632878363132477, "sampling/importance_sampling_ratio/mean": 1.007057547569275, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5731233917176723, "clip_ratio/low_mean": 0.033882785588502884, "clip_ratio/low_min": 0.033882785588502884, "clip_ratio/high_mean": 0.05512813315726817, "clip_ratio/high_max": 0.05512813315726817, "clip_ratio/region_mean": 0.08901091874577105, "reward_total_mean": 0.9578008651733398, "reward_meter_mean": 0.9578008651733398, "reward_meter_std": 0.05439894273877144, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9578008651733398, "reward_total_composite_std": 0.05439894273877144, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 650.0} {"timestamp_utc": "2026-04-11T20:46:16Z", "mode": "eval", "global_step": 650, "epoch": 0.025100401606425703, "eval_loss": NaN, "eval_runtime": 81.5104, "eval_samples_per_second": 1.276, "eval_steps_per_second": 0.159, "eval_num_tokens": 1408050.0, "eval_completions/mean_length": 229.43269230769232, "eval_completions/min_length": 59.38461538461539, "eval_completions/max_length": 433.15384615384613, "eval_completions/clipped_ratio": 0.028846153846153848, "eval_completions/mean_terminated_length": 220.6758258526142, "eval_completions/min_terminated_length": 59.38461538461539, "eval_completions/max_terminated_length": 416.6923076923077, "eval_rewards/meter/mean": 0.8344126389576838, "eval_rewards/meter/std": 0.3256990680327782, "eval_rewards/count_adherence/mean": 0.9108192599736727, "eval_rewards/count_adherence/std": 0.1330794313779244, "eval_rewards/arabic_clean/mean": 0.9615384615384616, "eval_rewards/arabic_clean/std": 0.0900012942460867, "eval_rewards/total_composite/mean": 0.7413886831356928, "eval_rewards/total_composite/std": 0.3471943082717749, "eval_reward": 0.7413886831356928, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.007470924049042738, "eval_sampling/sampling_logp_difference/max": 0.9992800859304575, "eval_sampling/importance_sampling_ratio/min": 0.3777222243639139, "eval_sampling/importance_sampling_ratio/mean": 1.0019580951103797, "eval_sampling/importance_sampling_ratio/max": 1.3485183532421405, "eval_entropy": 0.07029147646748103, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7413886831356928, "eval_reward_meter_mean": 0.8344126389576838, "eval_reward_meter_std": 0.3256990680327782, "eval_reward_count_adherence_mean": 0.9108192599736727, "eval_reward_count_adherence_std": 0.1330794313779244, "eval_reward_arabic_clean_mean": 0.9615384615384616, "eval_reward_arabic_clean_std": 0.0900012942460867, "eval_reward_total_composite_mean": 0.7413886831356928, "eval_reward_total_composite_std": 0.3471943082717749, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 650.0} {"timestamp_utc": "2026-04-11T20:46:24Z", "mode": "train", "global_step": 651, "epoch": 0.025139017608897127, "loss": 0.0656, "grad_norm": 4.527357578277588, "learning_rate": 8.03030303030303e-06, "num_tokens": 1409954.0, "completions/mean_length": 74.0, "completions/min_length": 67.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9963353872299194, "rewards/meter/std": 0.0011508456664159894, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9963353872299194, "rewards/total_composite/std": 0.0011508456664159894, "reward": 0.9963353872299194, "reward_std": 0.001150845200754702, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021704083308577538, "sampling/sampling_logp_difference/max": 0.80377197265625, "sampling/importance_sampling_ratio/min": 0.44763731956481934, "sampling/importance_sampling_ratio/mean": 1.0058751106262207, "sampling/importance_sampling_ratio/max": 1.7427269220352173, "entropy": 0.11899975594133139, "clip_ratio/low_mean": 0.016023859614506364, "clip_ratio/low_min": 0.016023859614506364, "clip_ratio/high_mean": 0.007119925343431532, "clip_ratio/high_max": 0.007119925343431532, "clip_ratio/region_mean": 0.023143784957937896, "reward_total_mean": 0.9963353872299194, "reward_meter_mean": 0.9963353872299194, "reward_meter_std": 0.0011508456664159894, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9963353872299194, "reward_total_composite_std": 0.0011508456664159894, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 651.0} {"timestamp_utc": "2026-04-11T20:46:31Z", "mode": "train", "global_step": 652, "epoch": 0.02517763361136855, "loss": 0.0253, "grad_norm": 1.6498056650161743, "learning_rate": 8.027272727272728e-06, "num_tokens": 1413183.0, "completions/mean_length": 199.625, "completions/min_length": 188.0, "completions/max_length": 210.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 199.625, "completions/min_terminated_length": 188.0, "completions/max_terminated_length": 210.0, "rewards/meter/mean": 0.9899384379386902, "rewards/meter/std": 0.006561415735632181, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9899384379386902, "rewards/total_composite/std": 0.006561415735632181, "reward": 0.9899384379386902, "reward_std": 0.006561421323567629, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014925537630915642, "sampling/sampling_logp_difference/max": 0.9635627269744873, "sampling/importance_sampling_ratio/min": 0.3815311789512634, "sampling/importance_sampling_ratio/mean": 0.9996187686920166, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09447363950312138, "clip_ratio/low_mean": 0.007814512820914388, "clip_ratio/low_min": 0.007814512820914388, "clip_ratio/high_mean": 0.014073623344302177, "clip_ratio/high_max": 0.014073623344302177, "clip_ratio/region_mean": 0.021888136165216565, "reward_total_mean": 0.9899384379386902, "reward_meter_mean": 0.9899384379386902, "reward_meter_std": 0.006561415735632181, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9899384379386902, "reward_total_composite_std": 0.006561415735632181, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 652.0} {"timestamp_utc": "2026-04-11T20:46:36Z", "mode": "train", "global_step": 653, "epoch": 0.025216249613839975, "loss": 0.0197, "grad_norm": 7.536975383758545, "learning_rate": 8.024242424242425e-06, "num_tokens": 1414867.0, "completions/mean_length": 56.5, "completions/min_length": 51.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9857240915298462, "rewards/meter/std": 0.012984703294932842, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9857240915298462, "rewards/total_composite/std": 0.012984703294932842, "reward": 0.9857240915298462, "reward_std": 0.01298470888286829, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03485972434282303, "sampling/sampling_logp_difference/max": 1.4822931289672852, "sampling/importance_sampling_ratio/min": 0.22711628675460815, "sampling/importance_sampling_ratio/mean": 1.002363920211792, "sampling/importance_sampling_ratio/max": 1.8299564123153687, "entropy": 0.1751969624310732, "clip_ratio/low_mean": 0.008196720853447914, "clip_ratio/low_min": 0.008196720853447914, "clip_ratio/high_mean": 0.016036794404499233, "clip_ratio/high_max": 0.016036794404499233, "clip_ratio/region_mean": 0.024233515257947147, "reward_total_mean": 0.9857240915298462, "reward_meter_mean": 0.9857240915298462, "reward_meter_std": 0.012984703294932842, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9857240915298462, "reward_total_composite_std": 0.012984703294932842, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 653.0} {"timestamp_utc": "2026-04-11T20:46:43Z", "mode": "train", "global_step": 654, "epoch": 0.0252548656163114, "loss": 0.0237, "grad_norm": 2.307868242263794, "learning_rate": 8.021212121212122e-06, "num_tokens": 1417868.0, "completions/mean_length": 201.125, "completions/min_length": 187.0, "completions/max_length": 210.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 201.125, "completions/min_terminated_length": 187.0, "completions/max_terminated_length": 210.0, "rewards/meter/mean": 0.9972577095031738, "rewards/meter/std": 0.0012061396846547723, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9224463105201721, "rewards/total_composite/std": 0.10306739062070847, "reward": 0.9224463105201721, "reward_std": 0.10306738317012787, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01445098128169775, "sampling/sampling_logp_difference/max": 1.5971088409423828, "sampling/importance_sampling_ratio/min": 0.20248107612133026, "sampling/importance_sampling_ratio/mean": 0.999078631401062, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07093879743479192, "clip_ratio/low_mean": 0.004807692545000464, "clip_ratio/low_min": 0.004807692545000464, "clip_ratio/high_mean": 0.0062819147715345025, "clip_ratio/high_max": 0.0062819147715345025, "clip_ratio/region_mean": 0.011089607316534966, "reward_total_mean": 0.9224463105201721, "reward_meter_mean": 0.9972577095031738, "reward_meter_std": 0.0012061396846547723, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9224463105201721, "reward_total_composite_std": 0.10306739062070847, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 654.0} {"timestamp_utc": "2026-04-11T20:46:52Z", "mode": "train", "global_step": 655, "epoch": 0.025293481618782823, "loss": -0.1672, "grad_norm": 1.5250226259231567, "learning_rate": 8.018181818181818e-06, "num_tokens": 1420763.0, "completions/mean_length": 235.875, "completions/min_length": 190.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 196.42857360839844, "completions/min_terminated_length": 190.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.9358644485473633, "rewards/meter/std": 0.07746383547782898, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.2828427255153656, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.7309386134147644, "rewards/total_composite/std": 0.45206886529922485, "reward": 0.7309386134147644, "reward_std": 0.45206889510154724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029249267652630806, "sampling/sampling_logp_difference/max": 2.088442802429199, "sampling/importance_sampling_ratio/min": 0.12387989461421967, "sampling/importance_sampling_ratio/mean": 1.0036450624465942, "sampling/importance_sampling_ratio/max": 1.6952005624771118, "entropy": 0.18520103115588427, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/high_mean": 0.014083926740568131, "clip_ratio/high_max": 0.014083926740568131, "clip_ratio/region_mean": 0.01600700366543606, "reward_total_mean": 0.7309386134147644, "reward_meter_mean": 0.9358644485473633, "reward_meter_std": 0.07746383547782898, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.2828427255153656, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.7309386134147644, "reward_total_composite_std": 0.45206886529922485, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 655.0} {"timestamp_utc": "2026-04-11T20:46:58Z", "mode": "train", "global_step": 656, "epoch": 0.025332097621254247, "loss": 0.0723, "grad_norm": 8.56203842163086, "learning_rate": 8.015151515151515e-06, "num_tokens": 1422556.0, "completions/mean_length": 72.125, "completions/min_length": 66.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.8617337346076965, "rewards/meter/std": 0.2223026156425476, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8030082583427429, "rewards/total_composite/std": 0.25798237323760986, "reward": 0.8030082583427429, "reward_std": 0.25798237323760986, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06043868884444237, "sampling/sampling_logp_difference/max": 1.3347139358520508, "sampling/importance_sampling_ratio/min": 0.2632334530353546, "sampling/importance_sampling_ratio/mean": 1.0141513347625732, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42434411123394966, "clip_ratio/low_mean": 0.02076531178317964, "clip_ratio/low_min": 0.02076531178317964, "clip_ratio/high_mean": 0.027805027668364346, "clip_ratio/high_max": 0.027805027668364346, "clip_ratio/region_mean": 0.04857033945154399, "reward_total_mean": 0.8030082583427429, "reward_meter_mean": 0.8617337346076965, "reward_meter_std": 0.2223026156425476, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8030082583427429, "reward_total_composite_std": 0.25798237323760986, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 656.0} {"timestamp_utc": "2026-04-11T20:47:03Z", "mode": "train", "global_step": 657, "epoch": 0.02537071362372567, "loss": 0.0205, "grad_norm": 4.103635787963867, "learning_rate": 8.012121212121214e-06, "num_tokens": 1424597.0, "completions/mean_length": 100.125, "completions/min_length": 94.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.125, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.960498034954071, "rewards/meter/std": 0.06911002844572067, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.960498034954071, "rewards/total_composite/std": 0.06911002844572067, "reward": 0.960498034954071, "reward_std": 0.06911001354455948, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040338192135095596, "sampling/sampling_logp_difference/max": 2.197718858718872, "sampling/importance_sampling_ratio/min": 0.11105620115995407, "sampling/importance_sampling_ratio/mean": 0.9979158043861389, "sampling/importance_sampling_ratio/max": 1.8716368675231934, "entropy": 0.1652716277167201, "clip_ratio/low_mean": 0.0036057692486792803, "clip_ratio/low_min": 0.0036057692486792803, "clip_ratio/high_mean": 0.03874593786895275, "clip_ratio/high_max": 0.03874593786895275, "clip_ratio/region_mean": 0.04235170711763203, "reward_total_mean": 0.960498034954071, "reward_meter_mean": 0.960498034954071, "reward_meter_std": 0.06911002844572067, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.960498034954071, "reward_total_composite_std": 0.06911002844572067, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 657.0} {"timestamp_utc": "2026-04-11T20:47:08Z", "mode": "train", "global_step": 658, "epoch": 0.025409329626197096, "loss": -0.0134, "grad_norm": 4.32810115814209, "learning_rate": 8.00909090909091e-06, "num_tokens": 1426499.0, "completions/mean_length": 62.75, "completions/min_length": 61.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9962197542190552, "rewards/meter/std": 0.0011040258686989546, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9962197542190552, "rewards/total_composite/std": 0.0011040258686989546, "reward": 0.9962197542190552, "reward_std": 0.0011040277313441038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014161055907607079, "sampling/sampling_logp_difference/max": 0.6505541801452637, "sampling/importance_sampling_ratio/min": 0.5217565894126892, "sampling/importance_sampling_ratio/mean": 1.0064129829406738, "sampling/importance_sampling_ratio/max": 1.8842612504959106, "entropy": 0.08390368055552244, "clip_ratio/low_mean": 0.006147540640085936, "clip_ratio/low_min": 0.006147540640085936, "clip_ratio/high_mean": 0.0056925995741039515, "clip_ratio/high_max": 0.0056925995741039515, "clip_ratio/region_mean": 0.011840140214189887, "reward_total_mean": 0.9962197542190552, "reward_meter_mean": 0.9962197542190552, "reward_meter_std": 0.0011040258686989546, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9962197542190552, "reward_total_composite_std": 0.0011040258686989546, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 658.0} {"timestamp_utc": "2026-04-11T20:47:18Z", "mode": "train", "global_step": 659, "epoch": 0.02544794562866852, "loss": -0.2596, "grad_norm": 0.485461562871933, "learning_rate": 8.006060606060607e-06, "num_tokens": 1428883.0, "completions/mean_length": 303.0, "completions/min_length": 170.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 177.60000610351562, "completions/min_terminated_length": 170.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.8517926931381226, "rewards/meter/std": 0.34845513105392456, "rewards/count_adherence/mean": 0.6749999523162842, "rewards/count_adherence/std": 0.41317588090896606, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6187050342559814, "rewards/total_composite/std": 0.47358113527297974, "reward": 0.6187050342559814, "reward_std": 0.4735811650753021, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009265605360269547, "sampling/sampling_logp_difference/max": 1.4901410341262817, "sampling/importance_sampling_ratio/min": 0.22534087300300598, "sampling/importance_sampling_ratio/mean": 0.9996970891952515, "sampling/importance_sampling_ratio/max": 1.5009973049163818, "entropy": 0.027310603065416217, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004928556620143354, "clip_ratio/high_max": 0.004928556620143354, "clip_ratio/region_mean": 0.004928556620143354, "reward_total_mean": 0.6187050342559814, "reward_meter_mean": 0.8517926931381226, "reward_meter_std": 0.34845513105392456, "reward_count_adherence_mean": 0.6749999523162842, "reward_count_adherence_std": 0.41317588090896606, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6187050342559814, "reward_total_composite_std": 0.47358113527297974, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 659.0} {"timestamp_utc": "2026-04-11T20:47:28Z", "mode": "train", "global_step": 660, "epoch": 0.025486561631139944, "loss": -0.0708, "grad_norm": 1.384261965751648, "learning_rate": 8.003030303030304e-06, "num_tokens": 1430370.0, "completions/mean_length": 213.875, "completions/min_length": 28.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.5319421887397766, "rewards/meter/std": 0.4762398600578308, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.5111215114593506, "rewards/total_composite/std": 0.4977530539035797, "reward": 0.5111215114593506, "reward_std": 0.4977530539035797, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09258230775594711, "sampling/sampling_logp_difference/max": 1.1651626825332642, "sampling/importance_sampling_ratio/min": 0.3118719160556793, "sampling/importance_sampling_ratio/mean": 1.0290229320526123, "sampling/importance_sampling_ratio/max": 1.871475338935852, "entropy": 0.5543557330965996, "clip_ratio/low_mean": 0.013392857275903225, "clip_ratio/low_min": 0.013392857275903225, "clip_ratio/high_mean": 0.04536097287200391, "clip_ratio/high_max": 0.04536097287200391, "clip_ratio/region_mean": 0.05875383014790714, "reward_total_mean": 0.5111215114593506, "reward_meter_mean": 0.5319421887397766, "reward_meter_std": 0.4762398600578308, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.5111215114593506, "reward_total_composite_std": 0.4977530539035797, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 660.0} {"timestamp_utc": "2026-04-11T20:47:37Z", "mode": "train", "global_step": 661, "epoch": 0.025525177633611368, "loss": -0.0815, "grad_norm": 6.0121989250183105, "learning_rate": 8.000000000000001e-06, "num_tokens": 1431872.0, "completions/mean_length": 111.75, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 54.57143020629883, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.7611011266708374, "rewards/meter/std": 0.4270692467689514, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7431202530860901, "rewards/total_composite/std": 0.45863598585128784, "reward": 0.7431202530860901, "reward_std": 0.45863595604896545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05083911120891571, "sampling/sampling_logp_difference/max": 2.2429542541503906, "sampling/importance_sampling_ratio/min": 0.10614446550607681, "sampling/importance_sampling_ratio/mean": 1.0131621360778809, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20556125091388822, "clip_ratio/low_mean": 0.006465517450124025, "clip_ratio/low_min": 0.006465517450124025, "clip_ratio/high_mean": 0.013890477130189538, "clip_ratio/high_max": 0.013890477130189538, "clip_ratio/region_mean": 0.020355994580313563, "reward_total_mean": 0.7431202530860901, "reward_meter_mean": 0.7611011266708374, "reward_meter_std": 0.4270692467689514, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7431202530860901, "reward_total_composite_std": 0.45863598585128784, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 661.0} {"timestamp_utc": "2026-04-11T20:47:42Z", "mode": "train", "global_step": 662, "epoch": 0.025563793636082792, "loss": 0.0131, "grad_norm": 3.3641293048858643, "learning_rate": 7.996969696969697e-06, "num_tokens": 1433580.0, "completions/mean_length": 56.5, "completions/min_length": 55.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9946731925010681, "rewards/meter/std": 0.00036217921297065914, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9946731925010681, "rewards/total_composite/std": 0.00036217921297065914, "reward": 0.9946731925010681, "reward_std": 0.0003621731011662632, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015676047652959824, "sampling/sampling_logp_difference/max": 1.2792699337005615, "sampling/importance_sampling_ratio/min": 0.2782403528690338, "sampling/importance_sampling_ratio/mean": 1.0013954639434814, "sampling/importance_sampling_ratio/max": 1.5647435188293457, "entropy": 0.11290217051282525, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.015325088053941727, "clip_ratio/high_max": 0.015325088053941727, "clip_ratio/region_mean": 0.019635432865470648, "reward_total_mean": 0.9946731925010681, "reward_meter_mean": 0.9946731925010681, "reward_meter_std": 0.00036217921297065914, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9946731925010681, "reward_total_composite_std": 0.00036217921297065914, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 662.0} {"timestamp_utc": "2026-04-11T20:47:52Z", "mode": "train", "global_step": 663, "epoch": 0.025602409638554216, "loss": 0.1892, "grad_norm": 3.8648812770843506, "learning_rate": 7.993939393939396e-06, "num_tokens": 1436063.0, "completions/mean_length": 191.375, "completions/min_length": 101.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 145.57144165039062, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 379.0, "rewards/meter/mean": 0.8358995318412781, "rewards/meter/std": 0.335735559463501, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.7419799566268921, "rewards/total_composite/std": 0.4580281972885132, "reward": 0.7419799566268921, "reward_std": 0.4580281972885132, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06357217580080032, "sampling/sampling_logp_difference/max": 1.2052602767944336, "sampling/importance_sampling_ratio/min": 0.3878151774406433, "sampling/importance_sampling_ratio/mean": 1.0154433250427246, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7347276965156198, "clip_ratio/low_mean": 0.004947229754179716, "clip_ratio/low_min": 0.004947229754179716, "clip_ratio/high_mean": 0.0069945488357916474, "clip_ratio/high_max": 0.0069945488357916474, "clip_ratio/region_mean": 0.011941778589971364, "reward_total_mean": 0.7419799566268921, "reward_meter_mean": 0.8358995318412781, "reward_meter_std": 0.335735559463501, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.7419799566268921, "reward_total_composite_std": 0.4580281972885132, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 663.0} {"timestamp_utc": "2026-04-11T20:48:02Z", "mode": "train", "global_step": 664, "epoch": 0.02564102564102564, "loss": -0.1597, "grad_norm": 0.7546610832214355, "learning_rate": 7.990909090909091e-06, "num_tokens": 1438281.0, "completions/mean_length": 156.25, "completions/min_length": 102.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 105.42857360839844, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.9944875240325928, "rewards/meter/std": 0.00466564204543829, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.2519763112068176, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8293492197990417, "rewards/total_composite/std": 0.25151386857032776, "reward": 0.8293492197990417, "reward_std": 0.25151386857032776, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006639350205659866, "sampling/sampling_logp_difference/max": 0.5072813034057617, "sampling/importance_sampling_ratio/min": 0.6021303534507751, "sampling/importance_sampling_ratio/mean": 1.0022079944610596, "sampling/importance_sampling_ratio/max": 1.4435498714447021, "entropy": 0.04206883814185858, "clip_ratio/low_mean": 0.004576730658300221, "clip_ratio/low_min": 0.004576730658300221, "clip_ratio/high_mean": 0.0023809524718672037, "clip_ratio/high_max": 0.0023809524718672037, "clip_ratio/region_mean": 0.006957683130167425, "reward_total_mean": 0.8293492197990417, "reward_meter_mean": 0.9944875240325928, "reward_meter_std": 0.00466564204543829, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.2519763112068176, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8293492197990417, "reward_total_composite_std": 0.25151386857032776, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 664.0} {"timestamp_utc": "2026-04-11T20:48:12Z", "mode": "train", "global_step": 665, "epoch": 0.025679641643497064, "loss": 0.009, "grad_norm": 2.3664374351501465, "learning_rate": 7.987878787878789e-06, "num_tokens": 1440140.0, "completions/mean_length": 200.375, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 96.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 232.0, "rewards/meter/mean": 0.9056166410446167, "rewards/meter/std": 0.1688472479581833, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.26726123690605164, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.43737679719924927, "rewards/total_composite/std": 0.4157540798187256, "reward": 0.43737679719924927, "reward_std": 0.4157540500164032, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09464051574468613, "sampling/sampling_logp_difference/max": 2.704573631286621, "sampling/importance_sampling_ratio/min": 0.06689883768558502, "sampling/importance_sampling_ratio/mean": 1.0214041471481323, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7333184946328402, "clip_ratio/low_mean": 0.007004310376942158, "clip_ratio/low_min": 0.007004310376942158, "clip_ratio/high_mean": 0.02338858088478446, "clip_ratio/high_max": 0.02338858088478446, "clip_ratio/region_mean": 0.030392891261726618, "reward_total_mean": 0.43737679719924927, "reward_meter_mean": 0.9056166410446167, "reward_meter_std": 0.1688472479581833, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.26726123690605164, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.43737679719924927, "reward_total_composite_std": 0.4157540798187256, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 665.0} {"timestamp_utc": "2026-04-11T20:48:22Z", "mode": "train", "global_step": 666, "epoch": 0.02571825764596849, "loss": -0.1506, "grad_norm": 0.872067391872406, "learning_rate": 7.984848484848486e-06, "num_tokens": 1441828.0, "completions/mean_length": 111.0, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 53.71428680419922, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9134105443954468, "rewards/meter/std": 0.2150774449110031, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8657701015472412, "rewards/total_composite/std": 0.34982457756996155, "reward": 0.8657701015472412, "reward_std": 0.34982454776763916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014446345157921314, "sampling/sampling_logp_difference/max": 0.7338418960571289, "sampling/importance_sampling_ratio/min": 0.4800611138343811, "sampling/importance_sampling_ratio/mean": 1.0026583671569824, "sampling/importance_sampling_ratio/max": 1.3767644166946411, "entropy": 0.09272160427644849, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.016470797825604677, "clip_ratio/high_max": 0.016470797825604677, "clip_ratio/region_mean": 0.016470797825604677, "reward_total_mean": 0.8657701015472412, "reward_meter_mean": 0.9134105443954468, "reward_meter_std": 0.2150774449110031, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8657701015472412, "reward_total_composite_std": 0.34982457756996155, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 666.0} {"timestamp_utc": "2026-04-11T20:48:32Z", "mode": "train", "global_step": 667, "epoch": 0.025756873648439912, "loss": -0.2112, "grad_norm": 1.130478858947754, "learning_rate": 7.981818181818183e-06, "num_tokens": 1444070.0, "completions/mean_length": 167.25, "completions/min_length": 91.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 118.00000762939453, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9783296585083008, "rewards/meter/std": 0.02812933176755905, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.703811764717102, "rewards/total_composite/std": 0.30978062748908997, "reward": 0.703811764717102, "reward_std": 0.30978062748908997, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03481154143810272, "sampling/sampling_logp_difference/max": 1.4749164581298828, "sampling/importance_sampling_ratio/min": 0.22879783809185028, "sampling/importance_sampling_ratio/mean": 0.9934218525886536, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11036482639610767, "clip_ratio/low_mean": 0.007231405004858971, "clip_ratio/low_min": 0.007231405004858971, "clip_ratio/high_mean": 0.029473521979525685, "clip_ratio/high_max": 0.029473521979525685, "clip_ratio/region_mean": 0.036704926984384656, "reward_total_mean": 0.703811764717102, "reward_meter_mean": 0.9783296585083008, "reward_meter_std": 0.02812933176755905, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.703811764717102, "reward_total_composite_std": 0.30978062748908997, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 667.0} {"timestamp_utc": "2026-04-11T20:48:42Z", "mode": "train", "global_step": 668, "epoch": 0.025795489650911337, "loss": -0.0551, "grad_norm": 1.911139726638794, "learning_rate": 7.978787878787879e-06, "num_tokens": 1445654.0, "completions/mean_length": 223.0, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 49.60000228881836, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.45732593536376953, "rewards/meter/std": 0.43902209401130676, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/total_composite/mean": 0.31035739183425903, "rewards/total_composite/std": 0.4300127625465393, "reward": 0.31035739183425903, "reward_std": 0.4300127923488617, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07701684534549713, "sampling/sampling_logp_difference/max": 1.9657402038574219, "sampling/importance_sampling_ratio/min": 0.1400521844625473, "sampling/importance_sampling_ratio/mean": 0.9934643507003784, "sampling/importance_sampling_ratio/max": 1.7072809934616089, "entropy": 0.23709848895668983, "clip_ratio/low_mean": 0.012450980255380273, "clip_ratio/low_min": 0.012450980255380273, "clip_ratio/high_mean": 0.035714286379516125, "clip_ratio/high_max": 0.035714286379516125, "clip_ratio/region_mean": 0.0481652666348964, "reward_total_mean": 0.31035739183425903, "reward_meter_mean": 0.45732593536376953, "reward_meter_std": 0.43902209401130676, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_total_composite_mean": 0.31035739183425903, "reward_total_composite_std": 0.4300127625465393, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 668.0} {"timestamp_utc": "2026-04-11T20:48:51Z", "mode": "train", "global_step": 669, "epoch": 0.02583410565338276, "loss": -0.1525, "grad_norm": 0.7202282547950745, "learning_rate": 7.975757575757576e-06, "num_tokens": 1447416.0, "completions/mean_length": 114.25, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.42857360839844, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9792252779006958, "rewards/meter/std": 0.013368271291255951, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8576850891113281, "rewards/total_composite/std": 0.3468036651611328, "reward": 0.8576850891113281, "reward_std": 0.3468036353588104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0371677502989769, "sampling/sampling_logp_difference/max": 1.2777891159057617, "sampling/importance_sampling_ratio/min": 0.2786526679992676, "sampling/importance_sampling_ratio/mean": 1.0027592182159424, "sampling/importance_sampling_ratio/max": 1.9761621952056885, "entropy": 0.16035151248797774, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.026967209530994296, "clip_ratio/high_max": 0.026967209530994296, "clip_ratio/region_mean": 0.026967209530994296, "reward_total_mean": 0.8576850891113281, "reward_meter_mean": 0.9792252779006958, "reward_meter_std": 0.013368271291255951, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8576850891113281, "reward_total_composite_std": 0.3468036651611328, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 669.0} {"timestamp_utc": "2026-04-11T20:48:58Z", "mode": "train", "global_step": 670, "epoch": 0.025872721655854185, "loss": 0.0069, "grad_norm": 1.325692892074585, "learning_rate": 7.972727272727273e-06, "num_tokens": 1450701.0, "completions/mean_length": 206.625, "completions/min_length": 204.0, "completions/max_length": 211.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 206.625, "completions/min_terminated_length": 204.0, "completions/max_terminated_length": 211.0, "rewards/meter/mean": 0.9958693385124207, "rewards/meter/std": 0.0006848756456747651, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9750892519950867, "rewards/total_composite/std": 0.0581388883292675, "reward": 0.9750892519950867, "reward_std": 0.0581388995051384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008226564154028893, "sampling/sampling_logp_difference/max": 3.809736490249634, "sampling/importance_sampling_ratio/min": 0.02215401642024517, "sampling/importance_sampling_ratio/mean": 0.997574508190155, "sampling/importance_sampling_ratio/max": 1.4390126466751099, "entropy": 0.024741128087043762, "clip_ratio/low_mean": 0.0005924170836806297, "clip_ratio/low_min": 0.0005924170836806297, "clip_ratio/high_mean": 0.004833849321585149, "clip_ratio/high_max": 0.004833849321585149, "clip_ratio/region_mean": 0.005426266405265778, "reward_total_mean": 0.9750892519950867, "reward_meter_mean": 0.9958693385124207, "reward_meter_std": 0.0006848756456747651, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9750892519950867, "reward_total_composite_std": 0.0581388883292675, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 670.0} {"timestamp_utc": "2026-04-11T20:49:08Z", "mode": "train", "global_step": 671, "epoch": 0.02591133765832561, "loss": -0.1705, "grad_norm": 0.26956939697265625, "learning_rate": 7.96969696969697e-06, "num_tokens": 1452401.0, "completions/mean_length": 123.5, "completions/min_length": 68.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.8713666200637817, "rewards/meter/std": 0.35208529233932495, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8713666200637817, "rewards/total_composite/std": 0.35208529233932495, "reward": 0.8713666200637817, "reward_std": 0.35208529233932495, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002195620210841298, "sampling/sampling_logp_difference/max": 0.13106314837932587, "sampling/importance_sampling_ratio/min": 0.9497882127761841, "sampling/importance_sampling_ratio/mean": 1.0017800331115723, "sampling/importance_sampling_ratio/max": 1.1400396823883057, "entropy": 0.0182401331840083, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8713666200637817, "reward_meter_mean": 0.8713666200637817, "reward_meter_std": 0.35208529233932495, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8713666200637817, "reward_total_composite_std": 0.35208529233932495, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 671.0} {"timestamp_utc": "2026-04-11T20:49:17Z", "mode": "train", "global_step": 672, "epoch": 0.025949953660797033, "loss": -0.2234, "grad_norm": 0.7195173501968384, "learning_rate": 7.966666666666668e-06, "num_tokens": 1454679.0, "completions/mean_length": 174.75, "completions/min_length": 121.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 126.5714340209961, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.8742517232894897, "rewards/meter/std": 0.3532509505748749, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8742516040802002, "rewards/total_composite/std": 0.35325127840042114, "reward": 0.8742516040802002, "reward_std": 0.35325127840042114, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013026879169046879, "sampling/sampling_logp_difference/max": 1.502593994140625, "sampling/importance_sampling_ratio/min": 0.2225521206855774, "sampling/importance_sampling_ratio/mean": 1.0038602352142334, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03888068813830614, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.009001564932987094, "clip_ratio/high_max": 0.009001564932987094, "clip_ratio/region_mean": 0.009001564932987094, "reward_total_mean": 0.8742516040802002, "reward_meter_mean": 0.8742517232894897, "reward_meter_std": 0.3532509505748749, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8742516040802002, "reward_total_composite_std": 0.35325127840042114, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 672.0} {"timestamp_utc": "2026-04-11T20:49:23Z", "mode": "train", "global_step": 673, "epoch": 0.025988569663268457, "loss": 0.1307, "grad_norm": 12.640143394470215, "learning_rate": 7.963636363636365e-06, "num_tokens": 1456221.0, "completions/mean_length": 53.75, "completions/min_length": 45.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.36327680945396423, "rewards/meter/std": 0.3090669512748718, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.36327680945396423, "rewards/total_composite/std": 0.3090669512748718, "reward": 0.36327680945396423, "reward_std": 0.30906692147254944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0732530802488327, "sampling/sampling_logp_difference/max": 1.3676308393478394, "sampling/importance_sampling_ratio/min": 0.4030386209487915, "sampling/importance_sampling_ratio/mean": 1.0164148807525635, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6730491854250431, "clip_ratio/low_mean": 0.027953717159107327, "clip_ratio/low_min": 0.027953717159107327, "clip_ratio/high_mean": 0.020361394621431828, "clip_ratio/high_max": 0.020361394621431828, "clip_ratio/region_mean": 0.048315111780539155, "reward_total_mean": 0.36327680945396423, "reward_meter_mean": 0.36327680945396423, "reward_meter_std": 0.3090669512748718, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.36327680945396423, "reward_total_composite_std": 0.3090669512748718, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 673.0} {"timestamp_utc": "2026-04-11T20:49:32Z", "mode": "train", "global_step": 674, "epoch": 0.02602718566573988, "loss": -0.0662, "grad_norm": 2.720024347305298, "learning_rate": 7.96060606060606e-06, "num_tokens": 1457745.0, "completions/mean_length": 178.5, "completions/min_length": 59.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 67.33333587646484, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.6429893970489502, "rewards/meter/std": 0.4537588953971863, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.3948606848716736, "rewards/total_composite/std": 0.4677083194255829, "reward": 0.3948606848716736, "reward_std": 0.4677083194255829, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04642193019390106, "sampling/sampling_logp_difference/max": 1.2240619659423828, "sampling/importance_sampling_ratio/min": 0.29403337836265564, "sampling/importance_sampling_ratio/mean": 1.0118073225021362, "sampling/importance_sampling_ratio/max": 1.6525382995605469, "entropy": 0.2077210769057274, "clip_ratio/low_mean": 0.021200980991125107, "clip_ratio/low_min": 0.021200980991125107, "clip_ratio/high_mean": 0.013706985395401716, "clip_ratio/high_max": 0.013706985395401716, "clip_ratio/region_mean": 0.03490796638652682, "reward_total_mean": 0.3948606848716736, "reward_meter_mean": 0.6429893970489502, "reward_meter_std": 0.4537588953971863, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.3948606848716736, "reward_total_composite_std": 0.4677083194255829, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 674.0} {"timestamp_utc": "2026-04-11T20:49:42Z", "mode": "train", "global_step": 675, "epoch": 0.026065801668211305, "loss": -0.1123, "grad_norm": 0.316235214471817, "learning_rate": 7.957575757575758e-06, "num_tokens": 1459359.0, "completions/mean_length": 93.75, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8713666200637817, "rewards/meter/std": 0.35208529233932495, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8713666200637817, "rewards/total_composite/std": 0.35208529233932495, "reward": 0.8713666200637817, "reward_std": 0.35208529233932495, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0034471447579562664, "sampling/sampling_logp_difference/max": 0.18926194310188293, "sampling/importance_sampling_ratio/min": 0.8275697231292725, "sampling/importance_sampling_ratio/mean": 1.0003942251205444, "sampling/importance_sampling_ratio/max": 1.0252037048339844, "entropy": 0.020127628231421113, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0036764706019312143, "clip_ratio/high_max": 0.0036764706019312143, "clip_ratio/region_mean": 0.0036764706019312143, "reward_total_mean": 0.8713666200637817, "reward_meter_mean": 0.8713666200637817, "reward_meter_std": 0.35208529233932495, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8713666200637817, "reward_total_composite_std": 0.35208529233932495, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 675.0} {"timestamp_utc": "2026-04-11T20:49:47Z", "mode": "train", "global_step": 676, "epoch": 0.02610441767068273, "loss": 0.0368, "grad_norm": 9.799997329711914, "learning_rate": 7.954545454545455e-06, "num_tokens": 1461366.0, "completions/mean_length": 95.875, "completions/min_length": 77.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.875, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.8650763034820557, "rewards/meter/std": 0.34486204385757446, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8650763034820557, "rewards/total_composite/std": 0.34486204385757446, "reward": 0.8650763034820557, "reward_std": 0.34486207365989685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03299599140882492, "sampling/sampling_logp_difference/max": 2.3715991973876953, "sampling/importance_sampling_ratio/min": 0.09333135187625885, "sampling/importance_sampling_ratio/mean": 0.9979630708694458, "sampling/importance_sampling_ratio/max": 1.9796196222305298, "entropy": 0.11586113506928086, "clip_ratio/low_mean": 0.002475247485563159, "clip_ratio/low_min": 0.002475247485563159, "clip_ratio/high_mean": 0.0264303496805951, "clip_ratio/high_max": 0.0264303496805951, "clip_ratio/region_mean": 0.02890559716615826, "reward_total_mean": 0.8650763034820557, "reward_meter_mean": 0.8650763034820557, "reward_meter_std": 0.34486204385757446, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8650763034820557, "reward_total_composite_std": 0.34486204385757446, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 676.0} {"timestamp_utc": "2026-04-11T20:49:57Z", "mode": "train", "global_step": 677, "epoch": 0.026143033673154153, "loss": -0.0822, "grad_norm": 2.2629449367523193, "learning_rate": 7.951515151515152e-06, "num_tokens": 1462965.0, "completions/mean_length": 162.875, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.4891268312931061, "rewards/meter/std": 0.3679125905036926, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.4783456325531006, "rewards/total_composite/std": 0.38300520181655884, "reward": 0.4783456325531006, "reward_std": 0.38300520181655884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06157706305384636, "sampling/sampling_logp_difference/max": 1.4290897846221924, "sampling/importance_sampling_ratio/min": 0.23952685296535492, "sampling/importance_sampling_ratio/mean": 1.0180469751358032, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21544138342142105, "clip_ratio/low_mean": 0.010336743667721748, "clip_ratio/low_min": 0.010336743667721748, "clip_ratio/high_mean": 0.016666667070239782, "clip_ratio/high_max": 0.016666667070239782, "clip_ratio/region_mean": 0.02700341073796153, "reward_total_mean": 0.4783456325531006, "reward_meter_mean": 0.4891268312931061, "reward_meter_std": 0.3679125905036926, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.4783456325531006, "reward_total_composite_std": 0.38300520181655884, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 677.0} {"timestamp_utc": "2026-04-11T20:50:02Z", "mode": "train", "global_step": 678, "epoch": 0.026181649675625578, "loss": 0.0152, "grad_norm": 4.746488094329834, "learning_rate": 7.948484848484848e-06, "num_tokens": 1464966.0, "completions/mean_length": 79.125, "completions/min_length": 76.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.125, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9967953562736511, "rewards/meter/std": 0.001241661375388503, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9967953562736511, "rewards/total_composite/std": 0.001241661375388503, "reward": 0.9967953562736511, "reward_std": 0.0012416671961545944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01968878135085106, "sampling/sampling_logp_difference/max": 1.0641870498657227, "sampling/importance_sampling_ratio/min": 0.3450082242488861, "sampling/importance_sampling_ratio/mean": 1.000454306602478, "sampling/importance_sampling_ratio/max": 1.5599312782287598, "entropy": 0.13109817123040557, "clip_ratio/low_mean": 0.0030505952890962362, "clip_ratio/low_min": 0.0030505952890962362, "clip_ratio/high_mean": 0.009496793965809047, "clip_ratio/high_max": 0.009496793965809047, "clip_ratio/region_mean": 0.012547389254905283, "reward_total_mean": 0.9967953562736511, "reward_meter_mean": 0.9967953562736511, "reward_meter_std": 0.001241661375388503, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9967953562736511, "reward_total_composite_std": 0.001241661375388503, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 678.0} {"timestamp_utc": "2026-04-11T20:50:07Z", "mode": "train", "global_step": 679, "epoch": 0.026220265678097, "loss": -0.0002, "grad_norm": 0.028259722515940666, "learning_rate": 7.945454545454547e-06, "num_tokens": 1467526.0, "completions/mean_length": 136.0, "completions/min_length": 136.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.0, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.995849609375, "rewards/meter/std": 5.816352313559037e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.995849609375, "rewards/total_composite/std": 5.816352313559037e-06, "reward": 0.995849609375, "reward_std": 5.8162650020676665e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0008136914693750441, "sampling/sampling_logp_difference/max": 0.33158349990844727, "sampling/importance_sampling_ratio/min": 0.7177862524986267, "sampling/importance_sampling_ratio/mean": 1.0002460479736328, "sampling/importance_sampling_ratio/max": 1.0455117225646973, "entropy": 0.004926680441712961, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.995849609375, "reward_meter_mean": 0.995849609375, "reward_meter_std": 5.816352313559037e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.995849609375, "reward_total_composite_std": 5.816352313559037e-06, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 679.0} {"timestamp_utc": "2026-04-11T20:50:12Z", "mode": "train", "global_step": 680, "epoch": 0.02625888168056843, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.942424242424242e-06, "num_tokens": 1469324.0, "completions/mean_length": 67.75, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9958475828170776, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9958475828170776, "rewards/total_composite/std": 0.0, "reward": 0.9958475828170776, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0036529083736240864, "sampling/sampling_logp_difference/max": 0.923851490020752, "sampling/importance_sampling_ratio/min": 0.39698711037635803, "sampling/importance_sampling_ratio/mean": 0.9990282654762268, "sampling/importance_sampling_ratio/max": 1.0604045391082764, "entropy": 0.00782407948281616, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9958475828170776, "reward_meter_mean": 0.9958475828170776, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9958475828170776, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 680.0} {"timestamp_utc": "2026-04-11T20:50:18Z", "mode": "train", "global_step": 681, "epoch": 0.026297497683039853, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.93939393939394e-06, "num_tokens": 1471892.0, "completions/mean_length": 136.0, "completions/min_length": 136.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.0, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.9958475828170776, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9958475828170776, "rewards/total_composite/std": 0.0, "reward": 0.9958475828170776, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.000283001980278641, "sampling/sampling_logp_difference/max": 0.011258913204073906, "sampling/importance_sampling_ratio/min": 0.9975233674049377, "sampling/importance_sampling_ratio/mean": 1.000266671180725, "sampling/importance_sampling_ratio/max": 1.0113224983215332, "entropy": 0.0025078106846194714, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9958475828170776, "reward_meter_mean": 0.9958475828170776, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9958475828170776, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 681.0} {"timestamp_utc": "2026-04-11T20:50:23Z", "mode": "train", "global_step": 682, "epoch": 0.026336113685511277, "loss": -0.0718, "grad_norm": 6.921201705932617, "learning_rate": 7.936363636363637e-06, "num_tokens": 1473998.0, "completions/mean_length": 98.25, "completions/min_length": 77.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.25, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9740473031997681, "rewards/meter/std": 0.05396129563450813, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.939018964767456, "rewards/total_composite/std": 0.15298905968666077, "reward": 0.939018964767456, "reward_std": 0.15298905968666077, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033082515001297, "sampling/sampling_logp_difference/max": 2.2817556858062744, "sampling/importance_sampling_ratio/min": 0.1021047830581665, "sampling/importance_sampling_ratio/mean": 0.9985213279724121, "sampling/importance_sampling_ratio/max": 1.7105910778045654, "entropy": 0.16102673672139645, "clip_ratio/low_mean": 0.008116883225739002, "clip_ratio/low_min": 0.008116883225739002, "clip_ratio/high_mean": 0.013591251568868756, "clip_ratio/high_max": 0.013591251568868756, "clip_ratio/region_mean": 0.02170813479460776, "reward_total_mean": 0.939018964767456, "reward_meter_mean": 0.9740473031997681, "reward_meter_std": 0.05396129563450813, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.939018964767456, "reward_total_composite_std": 0.15298905968666077, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 682.0} {"timestamp_utc": "2026-04-11T20:50:28Z", "mode": "train", "global_step": 683, "epoch": 0.0263747296879827, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.933333333333334e-06, "num_tokens": 1475894.0, "completions/mean_length": 68.0, "completions/min_length": 68.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9958475828170776, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9958475828170776, "rewards/total_composite/std": 0.0, "reward": 0.9958475828170776, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006028310745023191, "sampling/sampling_logp_difference/max": 0.042691268026828766, "sampling/importance_sampling_ratio/min": 0.9582071304321289, "sampling/importance_sampling_ratio/mean": 1.0003547668457031, "sampling/importance_sampling_ratio/max": 1.0147513151168823, "entropy": 0.005274175782687962, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9958475828170776, "reward_meter_mean": 0.9958475828170776, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9958475828170776, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 683.0} {"timestamp_utc": "2026-04-11T20:50:38Z", "mode": "train", "global_step": 684, "epoch": 0.026413345690454126, "loss": -0.2422, "grad_norm": 0.17992635071277618, "learning_rate": 7.930303030303031e-06, "num_tokens": 1478337.0, "completions/mean_length": 202.375, "completions/min_length": 157.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 158.1428680419922, "completions/min_terminated_length": 157.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9706461429595947, "rewards/meter/std": 0.07604657858610153, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8728405833244324, "rewards/total_composite/std": 0.3526812195777893, "reward": 0.8728405833244324, "reward_std": 0.3526812195777893, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004087422508746386, "sampling/sampling_logp_difference/max": 0.8292655944824219, "sampling/importance_sampling_ratio/min": 0.4363696575164795, "sampling/importance_sampling_ratio/mean": 1.0004380941390991, "sampling/importance_sampling_ratio/max": 1.2241318225860596, "entropy": 0.02341610216535628, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.003926091885659844, "clip_ratio/high_max": 0.003926091885659844, "clip_ratio/region_mean": 0.003926091885659844, "reward_total_mean": 0.8728405833244324, "reward_meter_mean": 0.9706461429595947, "reward_meter_std": 0.07604657858610153, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8728405833244324, "reward_total_composite_std": 0.3526812195777893, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 684.0} {"timestamp_utc": "2026-04-11T20:50:43Z", "mode": "train", "global_step": 685, "epoch": 0.02645196169292555, "loss": -0.0018, "grad_norm": 4.395933151245117, "learning_rate": 7.927272727272729e-06, "num_tokens": 1480159.0, "completions/mean_length": 68.75, "completions/min_length": 36.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.835988461971283, "rewards/meter/std": 0.336283802986145, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7773693203926086, "rewards/total_composite/std": 0.35625946521759033, "reward": 0.7773693203926086, "reward_std": 0.35625946521759033, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02176571451127529, "sampling/sampling_logp_difference/max": 1.824343204498291, "sampling/importance_sampling_ratio/min": 0.3406663239002228, "sampling/importance_sampling_ratio/mean": 1.001936912536621, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1104641705751419, "clip_ratio/low_mean": 0.009424603311344981, "clip_ratio/low_min": 0.009424603311344981, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.009424603311344981, "reward_total_mean": 0.7773693203926086, "reward_meter_mean": 0.835988461971283, "reward_meter_std": 0.336283802986145, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7773693203926086, "reward_total_composite_std": 0.35625946521759033, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 685.0} {"timestamp_utc": "2026-04-11T20:50:52Z", "mode": "train", "global_step": 686, "epoch": 0.026490577695396974, "loss": -0.0681, "grad_norm": 3.5272436141967773, "learning_rate": 7.924242424242426e-06, "num_tokens": 1481726.0, "completions/mean_length": 96.875, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.57143020629883, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.6868791580200195, "rewards/meter/std": 0.431110143661499, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6857706308364868, "rewards/total_composite/std": 0.4331094026565552, "reward": 0.6857706308364868, "reward_std": 0.4331094026565552, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08996278792619705, "sampling/sampling_logp_difference/max": 3.1178348064422607, "sampling/importance_sampling_ratio/min": 0.04425288364291191, "sampling/importance_sampling_ratio/mean": 0.9939408302307129, "sampling/importance_sampling_ratio/max": 1.7951385974884033, "entropy": 0.4435567706823349, "clip_ratio/low_mean": 0.009615384973585606, "clip_ratio/low_min": 0.009615384973585606, "clip_ratio/high_mean": 0.06777206622064114, "clip_ratio/high_max": 0.06777206622064114, "clip_ratio/region_mean": 0.07738745119422674, "reward_total_mean": 0.6857706308364868, "reward_meter_mean": 0.6868791580200195, "reward_meter_std": 0.431110143661499, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6857706308364868, "reward_total_composite_std": 0.4331094026565552, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 686.0} {"timestamp_utc": "2026-04-11T20:51:02Z", "mode": "train", "global_step": 687, "epoch": 0.026529193697868398, "loss": -0.1703, "grad_norm": 1.5006870031356812, "learning_rate": 7.921212121212122e-06, "num_tokens": 1483541.0, "completions/mean_length": 129.875, "completions/min_length": 71.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 75.28572082519531, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8263435363769531, "rewards/meter/std": 0.34940168261528015, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.826325535774231, "rewards/total_composite/std": 0.34945032000541687, "reward": 0.826325535774231, "reward_std": 0.3494502902030945, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04448412358760834, "sampling/sampling_logp_difference/max": 1.331322193145752, "sampling/importance_sampling_ratio/min": 0.26412779092788696, "sampling/importance_sampling_ratio/mean": 1.0122401714324951, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.27454743906855583, "clip_ratio/low_mean": 0.0032051282469183207, "clip_ratio/low_min": 0.0032051282469183207, "clip_ratio/high_mean": 0.02495315601117909, "clip_ratio/high_max": 0.02495315601117909, "clip_ratio/region_mean": 0.02815828425809741, "reward_total_mean": 0.826325535774231, "reward_meter_mean": 0.8263435363769531, "reward_meter_std": 0.34940168261528015, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.826325535774231, "reward_total_composite_std": 0.34945032000541687, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 687.0} {"timestamp_utc": "2026-04-11T20:51:12Z", "mode": "train", "global_step": 688, "epoch": 0.026567809700339822, "loss": -0.1843, "grad_norm": 0.3745138645172119, "learning_rate": 7.918181818181819e-06, "num_tokens": 1485626.0, "completions/mean_length": 257.625, "completions/min_length": 103.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 105.0, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.6885502934455872, "rewards/meter/std": 0.44270819425582886, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.36443448066711426, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.618360161781311, "rewards/total_composite/std": 0.5120509266853333, "reward": 0.618360161781311, "reward_std": 0.5120508670806885, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004873558413237333, "sampling/sampling_logp_difference/max": 0.44519710540771484, "sampling/importance_sampling_ratio/min": 0.6406980156898499, "sampling/importance_sampling_ratio/mean": 0.9994492530822754, "sampling/importance_sampling_ratio/max": 1.351849913597107, "entropy": 0.016713016433641315, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.003572082845494151, "clip_ratio/high_max": 0.003572082845494151, "clip_ratio/region_mean": 0.003572082845494151, "reward_total_mean": 0.618360161781311, "reward_meter_mean": 0.6885502934455872, "reward_meter_std": 0.44270819425582886, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.36443448066711426, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.618360161781311, "reward_total_composite_std": 0.5120509266853333, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 688.0} {"timestamp_utc": "2026-04-11T20:51:22Z", "mode": "train", "global_step": 689, "epoch": 0.026606425702811246, "loss": -0.2404, "grad_norm": 0.6602917313575745, "learning_rate": 7.915151515151516e-06, "num_tokens": 1489295.0, "completions/mean_length": 312.625, "completions/min_length": 271.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 284.14288330078125, "completions/min_terminated_length": 271.0, "completions/max_terminated_length": 289.0, "rewards/meter/mean": 0.7451716661453247, "rewards/meter/std": 0.35534098744392395, "rewards/count_adherence/mean": 0.7678571939468384, "rewards/count_adherence/std": 0.3142625391483307, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6504477858543396, "rewards/total_composite/std": 0.30304864048957825, "reward": 0.6504477858543396, "reward_std": 0.30304864048957825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011496911756694317, "sampling/sampling_logp_difference/max": 1.5666828155517578, "sampling/importance_sampling_ratio/min": 0.20873644948005676, "sampling/importance_sampling_ratio/mean": 1.0029361248016357, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.058850419241935015, "clip_ratio/low_mean": 0.0012975778663530946, "clip_ratio/low_min": 0.0012975778663530946, "clip_ratio/high_mean": 0.005277197604300454, "clip_ratio/high_max": 0.005277197604300454, "clip_ratio/region_mean": 0.006574775470653549, "reward_total_mean": 0.6504477858543396, "reward_meter_mean": 0.7451716661453247, "reward_meter_std": 0.35534098744392395, "reward_count_adherence_mean": 0.7678571939468384, "reward_count_adherence_std": 0.3142625391483307, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6504477858543396, "reward_total_composite_std": 0.30304864048957825, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 689.0} {"timestamp_utc": "2026-04-11T20:51:27Z", "mode": "train", "global_step": 690, "epoch": 0.02664504170528267, "loss": 0.0113, "grad_norm": 3.4013071060180664, "learning_rate": 7.912121212121213e-06, "num_tokens": 1491205.0, "completions/mean_length": 72.75, "completions/min_length": 69.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.75, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9906994104385376, "rewards/meter/std": 0.004620725754648447, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9906994104385376, "rewards/total_composite/std": 0.004620725754648447, "reward": 0.9906994104385376, "reward_std": 0.004620728548616171, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019962970167398453, "sampling/sampling_logp_difference/max": 0.8946304321289062, "sampling/importance_sampling_ratio/min": 0.40875864028930664, "sampling/importance_sampling_ratio/mean": 1.0015352964401245, "sampling/importance_sampling_ratio/max": 1.627432107925415, "entropy": 0.08100668899714947, "clip_ratio/low_mean": 0.009722222341224551, "clip_ratio/low_min": 0.009722222341224551, "clip_ratio/high_mean": 0.007072941050864756, "clip_ratio/high_max": 0.007072941050864756, "clip_ratio/region_mean": 0.016795163392089307, "reward_total_mean": 0.9906994104385376, "reward_meter_mean": 0.9906994104385376, "reward_meter_std": 0.004620725754648447, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9906994104385376, "reward_total_composite_std": 0.004620725754648447, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 690.0} {"timestamp_utc": "2026-04-11T20:51:36Z", "mode": "train", "global_step": 691, "epoch": 0.026683657707754094, "loss": 0.0004, "grad_norm": 0.15184305608272552, "learning_rate": 7.909090909090909e-06, "num_tokens": 1495958.0, "completions/mean_length": 374.125, "completions/min_length": 374.0, "completions/max_length": 375.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 374.125, "completions/min_terminated_length": 374.0, "completions/max_terminated_length": 375.0, "rewards/meter/mean": 0.9958058595657349, "rewards/meter/std": 0.00011801117943832651, "rewards/count_adherence/mean": 0.8461538553237915, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8426049947738647, "rewards/total_composite/std": 9.986695658881217e-05, "reward": 0.8426049947738647, "reward_std": 9.987598605221137e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0013432148844003677, "sampling/sampling_logp_difference/max": 1.4840025901794434, "sampling/importance_sampling_ratio/min": 0.22672836482524872, "sampling/importance_sampling_ratio/mean": 0.9994156956672668, "sampling/importance_sampling_ratio/max": 1.0696345567703247, "entropy": 0.0032385573722422123, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.000668449210934341, "clip_ratio/high_max": 0.000668449210934341, "clip_ratio/region_mean": 0.000668449210934341, "reward_total_mean": 0.8426049947738647, "reward_meter_mean": 0.9958058595657349, "reward_meter_std": 0.00011801117943832651, "reward_count_adherence_mean": 0.8461538553237915, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8426049947738647, "reward_total_composite_std": 9.986695658881217e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 691.0} {"timestamp_utc": "2026-04-11T20:51:41Z", "mode": "train", "global_step": 692, "epoch": 0.02672227371022552, "loss": 0.0283, "grad_norm": 8.2815580368042, "learning_rate": 7.906060606060608e-06, "num_tokens": 1497859.0, "completions/mean_length": 79.625, "completions/min_length": 77.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.625, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.8718336224555969, "rewards/meter/std": 0.3495596945285797, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8718336224555969, "rewards/total_composite/std": 0.3495596945285797, "reward": 0.8718336224555969, "reward_std": 0.3495596945285797, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022529320791363716, "sampling/sampling_logp_difference/max": 0.8825089931488037, "sampling/importance_sampling_ratio/min": 0.4137435555458069, "sampling/importance_sampling_ratio/mean": 1.0064815282821655, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1280956557020545, "clip_ratio/low_mean": 0.007352941203862429, "clip_ratio/low_min": 0.007352941203862429, "clip_ratio/high_mean": 0.015544294146820903, "clip_ratio/high_max": 0.015544294146820903, "clip_ratio/region_mean": 0.02289723535068333, "reward_total_mean": 0.8718336224555969, "reward_meter_mean": 0.8718336224555969, "reward_meter_std": 0.3495596945285797, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8718336224555969, "reward_total_composite_std": 0.3495596945285797, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 692.0} {"timestamp_utc": "2026-04-11T20:51:46Z", "mode": "train", "global_step": 693, "epoch": 0.026760889712696943, "loss": 0.0015, "grad_norm": 2.438490152359009, "learning_rate": 7.903030303030303e-06, "num_tokens": 1499613.0, "completions/mean_length": 64.25, "completions/min_length": 64.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9994999170303345, "rewards/meter/std": 6.49320863885805e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9994999170303345, "rewards/total_composite/std": 6.49320863885805e-05, "reward": 0.9994999170303345, "reward_std": 6.491857493529096e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020598432049155235, "sampling/sampling_logp_difference/max": 0.7588024139404297, "sampling/importance_sampling_ratio/min": 0.4682268500328064, "sampling/importance_sampling_ratio/mean": 1.0003265142440796, "sampling/importance_sampling_ratio/max": 1.931307315826416, "entropy": 0.10410524718463421, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.011658653849735856, "clip_ratio/high_max": 0.011658653849735856, "clip_ratio/region_mean": 0.015564903849735856, "reward_total_mean": 0.9994999170303345, "reward_meter_mean": 0.9994999170303345, "reward_meter_std": 6.49320863885805e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9994999170303345, "reward_total_composite_std": 6.49320863885805e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 693.0} {"timestamp_utc": "2026-04-11T20:51:51Z", "mode": "train", "global_step": 694, "epoch": 0.026799505715168367, "loss": 0.0212, "grad_norm": 3.0967864990234375, "learning_rate": 7.9e-06, "num_tokens": 1501923.0, "completions/mean_length": 107.75, "completions/min_length": 105.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.75, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9867300987243652, "rewards/meter/std": 0.011255268007516861, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9867300987243652, "rewards/total_composite/std": 0.011255268007516861, "reward": 0.9867300987243652, "reward_std": 0.011255279183387756, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01411415170878172, "sampling/sampling_logp_difference/max": 2.3374826908111572, "sampling/importance_sampling_ratio/min": 0.09657042473554611, "sampling/importance_sampling_ratio/mean": 1.0007671117782593, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06845425767824054, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.0045688546961173415, "clip_ratio/high_max": 0.0045688546961173415, "clip_ratio/region_mean": 0.007947233156301081, "reward_total_mean": 0.9867300987243652, "reward_meter_mean": 0.9867300987243652, "reward_meter_std": 0.011255268007516861, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9867300987243652, "reward_total_composite_std": 0.011255268007516861, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 694.0} {"timestamp_utc": "2026-04-11T20:51:57Z", "mode": "train", "global_step": 695, "epoch": 0.02683812171763979, "loss": 0.0006, "grad_norm": 1.181228518486023, "learning_rate": 7.896969696969698e-06, "num_tokens": 1504596.0, "completions/mean_length": 132.125, "completions/min_length": 132.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.125, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9894136190414429, "rewards/meter/std": 3.8669753848807886e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9894136190414429, "rewards/total_composite/std": 3.8669753848807886e-05, "reward": 0.9894136190414429, "reward_std": 3.867876512231305e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0022934156004339457, "sampling/sampling_logp_difference/max": 1.3005752563476562, "sampling/importance_sampling_ratio/min": 0.27237507700920105, "sampling/importance_sampling_ratio/mean": 0.9996275305747986, "sampling/importance_sampling_ratio/max": 1.0533556938171387, "entropy": 0.010945796559099108, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9894136190414429, "reward_meter_mean": 0.9894136190414429, "reward_meter_std": 3.8669753848807886e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9894136190414429, "reward_total_composite_std": 3.8669753848807886e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 695.0} {"timestamp_utc": "2026-04-11T20:52:01Z", "mode": "train", "global_step": 696, "epoch": 0.026876737720111215, "loss": 0.0778, "grad_norm": 7.622135639190674, "learning_rate": 7.893939393939395e-06, "num_tokens": 1506380.0, "completions/mean_length": 55.0, "completions/min_length": 51.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.987579345703125, "rewards/meter/std": 0.006086904555559158, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.987579345703125, "rewards/total_composite/std": 0.006086904555559158, "reward": 0.987579345703125, "reward_std": 0.006086910609155893, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020231839269399643, "sampling/sampling_logp_difference/max": 0.8066816329956055, "sampling/importance_sampling_ratio/min": 0.4463367462158203, "sampling/importance_sampling_ratio/mean": 1.0072104930877686, "sampling/importance_sampling_ratio/max": 1.7266788482666016, "entropy": 0.11908045504242182, "clip_ratio/low_mean": 0.009615384973585606, "clip_ratio/low_min": 0.009615384973585606, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.009615384973585606, "reward_total_mean": 0.987579345703125, "reward_meter_mean": 0.987579345703125, "reward_meter_std": 0.006086904555559158, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.987579345703125, "reward_total_composite_std": 0.006086904555559158, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 696.0} {"timestamp_utc": "2026-04-11T20:52:11Z", "mode": "train", "global_step": 697, "epoch": 0.02691535372258264, "loss": -0.2243, "grad_norm": 0.5127809047698975, "learning_rate": 7.89090909090909e-06, "num_tokens": 1508829.0, "completions/mean_length": 177.125, "completions/min_length": 120.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 129.2857208251953, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.9192803502082825, "rewards/meter/std": 0.22436398267745972, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8737781047821045, "rewards/total_composite/std": 0.3530622720718384, "reward": 0.8737781047821045, "reward_std": 0.3530622720718384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01873595081269741, "sampling/sampling_logp_difference/max": 1.4939966201782227, "sampling/importance_sampling_ratio/min": 0.2244737297296524, "sampling/importance_sampling_ratio/mean": 1.0008881092071533, "sampling/importance_sampling_ratio/max": 1.5339782238006592, "entropy": 0.06667920062318444, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.017259016167372465, "clip_ratio/high_max": 0.017259016167372465, "clip_ratio/region_mean": 0.017259016167372465, "reward_total_mean": 0.8737781047821045, "reward_meter_mean": 0.9192803502082825, "reward_meter_std": 0.22436398267745972, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8737781047821045, "reward_total_composite_std": 0.3530622720718384, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 697.0} {"timestamp_utc": "2026-04-11T20:52:21Z", "mode": "train", "global_step": 698, "epoch": 0.026953969725054063, "loss": -0.0113, "grad_norm": 0.6022924184799194, "learning_rate": 7.88787878787879e-06, "num_tokens": 1514518.0, "completions/mean_length": 461.125, "completions/min_length": 459.0, "completions/max_length": 476.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 461.125, "completions/min_terminated_length": 459.0, "completions/max_terminated_length": 476.0, "rewards/meter/mean": 0.9955857992172241, "rewards/meter/std": 0.00036126485792919993, "rewards/count_adherence/mean": 0.6907894611358643, "rewards/count_adherence/std": 0.01860806532204151, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6877419948577881, "rewards/total_composite/std": 0.018604660406708717, "reward": 0.6877419948577881, "reward_std": 0.018604662269353867, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0022278090473264456, "sampling/sampling_logp_difference/max": 2.120715618133545, "sampling/importance_sampling_ratio/min": 0.11994575709104538, "sampling/importance_sampling_ratio/mean": 0.9998607039451599, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0036650724650826305, "clip_ratio/low_mean": 0.0010893245926126838, "clip_ratio/low_min": 0.0010893245926126838, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0010893245926126838, "reward_total_mean": 0.6877419948577881, "reward_meter_mean": 0.9955857992172241, "reward_meter_std": 0.00036126485792919993, "reward_count_adherence_mean": 0.6907894611358643, "reward_count_adherence_std": 0.01860806532204151, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6877419948577881, "reward_total_composite_std": 0.018604660406708717, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 698.0} {"timestamp_utc": "2026-04-11T20:52:26Z", "mode": "train", "global_step": 699, "epoch": 0.026992585727525487, "loss": 0.0053, "grad_norm": 8.910929679870605, "learning_rate": 7.884848484848485e-06, "num_tokens": 1516460.0, "completions/mean_length": 76.75, "completions/min_length": 75.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.75, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9973307847976685, "rewards/meter/std": 0.000212467581150122, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9973307847976685, "rewards/total_composite/std": 0.000212467581150122, "reward": 0.9973307847976685, "reward_std": 0.00021246756659820676, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007395791355520487, "sampling/sampling_logp_difference/max": 0.26826560497283936, "sampling/importance_sampling_ratio/min": 0.7647046446800232, "sampling/importance_sampling_ratio/mean": 1.003684639930725, "sampling/importance_sampling_ratio/max": 1.2306416034698486, "entropy": 0.058179288636893034, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9973307847976685, "reward_meter_mean": 0.9973307847976685, "reward_meter_std": 0.000212467581150122, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9973307847976685, "reward_total_composite_std": 0.000212467581150122, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 699.0} {"timestamp_utc": "2026-04-11T20:52:36Z", "mode": "train", "global_step": 700, "epoch": 0.02703120172999691, "loss": -0.2817, "grad_norm": 0.48487570881843567, "learning_rate": 7.881818181818182e-06, "num_tokens": 1520155.0, "completions/mean_length": 350.875, "completions/min_length": 297.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 327.8571472167969, "completions/min_terminated_length": 297.0, "completions/max_terminated_length": 352.0, "rewards/meter/mean": 0.9149429798126221, "rewards/meter/std": 0.22621573507785797, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.04629101976752281, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8083562850952148, "rewards/total_composite/std": 0.3296443521976471, "reward": 0.8083562850952148, "reward_std": 0.3296443521976471, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009414694271981716, "sampling/sampling_logp_difference/max": 1.6499791145324707, "sampling/importance_sampling_ratio/min": 0.1920539140701294, "sampling/importance_sampling_ratio/mean": 1.0025815963745117, "sampling/importance_sampling_ratio/max": 1.849985122680664, "entropy": 0.04480147338472307, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0049777484382502735, "clip_ratio/high_max": 0.0049777484382502735, "clip_ratio/region_mean": 0.0049777484382502735, "reward_total_mean": 0.8083562850952148, "reward_meter_mean": 0.9149429798126221, "reward_meter_std": 0.22621573507785797, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.04629101976752281, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8083562850952148, "reward_total_composite_std": 0.3296443521976471, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 700.0} {"timestamp_utc": "2026-04-11T20:54:04Z", "mode": "eval", "global_step": 700, "epoch": 0.02703120172999691, "eval_loss": NaN, "eval_runtime": 87.7965, "eval_samples_per_second": 1.185, "eval_steps_per_second": 0.148, "eval_num_tokens": 1520155.0, "eval_completions/mean_length": 255.04807692307693, "eval_completions/min_length": 59.69230769230769, "eval_completions/max_length": 473.2307692307692, "eval_completions/clipped_ratio": 0.11538461538461539, "eval_completions/mean_terminated_length": 221.03297189565805, "eval_completions/min_terminated_length": 59.69230769230769, "eval_completions/max_terminated_length": 410.84615384615387, "eval_rewards/meter/mean": 0.8046270196254437, "eval_rewards/meter/std": 0.33371496146831375, "eval_rewards/count_adherence/mean": 0.8984932028330289, "eval_rewards/count_adherence/std": 0.15058123836150536, "eval_rewards/arabic_clean/mean": 0.9134615384615384, "eval_rewards/arabic_clean/std": 0.1884146401515374, "eval_rewards/total_composite/mean": 0.7145874500274658, "eval_rewards/total_composite/std": 0.35720425844192505, "eval_reward": 0.7145874500274658, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.004972276931556945, "eval_sampling/sampling_logp_difference/max": 0.7181376493894137, "eval_sampling/importance_sampling_ratio/min": 0.5038195458742288, "eval_sampling/importance_sampling_ratio/mean": 1.001306639267848, "eval_sampling/importance_sampling_ratio/max": 1.2713435888290405, "eval_entropy": 0.04220953132384098, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7145874500274658, "eval_reward_meter_mean": 0.8046270196254437, "eval_reward_meter_std": 0.33371496146831375, "eval_reward_count_adherence_mean": 0.8984932028330289, "eval_reward_count_adherence_std": 0.15058123836150536, "eval_reward_arabic_clean_mean": 0.9134615384615384, "eval_reward_arabic_clean_std": 0.1884146401515374, "eval_reward_total_composite_mean": 0.7145874500274658, "eval_reward_total_composite_std": 0.35720425844192505, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 700.0} {"timestamp_utc": "2026-04-11T20:54:16Z", "mode": "train", "global_step": 701, "epoch": 0.027069817732468335, "loss": -0.185, "grad_norm": 0.23816236853599548, "learning_rate": 7.87878787878788e-06, "num_tokens": 1522123.0, "completions/mean_length": 134.0, "completions/min_length": 80.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9206724166870117, "rewards/meter/std": 0.19483156502246857, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8658612966537476, "rewards/total_composite/std": 0.349860817193985, "reward": 0.8658612966537476, "reward_std": 0.349860817193985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.001268567517399788, "sampling/sampling_logp_difference/max": 0.09954339265823364, "sampling/importance_sampling_ratio/min": 0.947121262550354, "sampling/importance_sampling_ratio/mean": 1.0009052753448486, "sampling/importance_sampling_ratio/max": 1.1046663522720337, "entropy": 0.014391513424925506, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8658612966537476, "reward_meter_mean": 0.9206724166870117, "reward_meter_std": 0.19483156502246857, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8658612966537476, "reward_total_composite_std": 0.349860817193985, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 701.0} {"timestamp_utc": "2026-04-11T20:54:26Z", "mode": "train", "global_step": 702, "epoch": 0.02710843373493976, "loss": -0.252, "grad_norm": 0.14842967689037323, "learning_rate": 7.875757575757577e-06, "num_tokens": 1524788.0, "completions/mean_length": 223.125, "completions/min_length": 181.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 181.85714721679688, "completions/min_terminated_length": 181.0, "completions/max_terminated_length": 184.0, "rewards/meter/mean": 0.9632084965705872, "rewards/meter/std": 0.07406298071146011, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.2357022613286972, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7214329242706299, "rewards/total_composite/std": 0.2915029525756836, "reward": 0.7214329242706299, "reward_std": 0.2915029525756836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.001055261236615479, "sampling/sampling_logp_difference/max": 0.2584630846977234, "sampling/importance_sampling_ratio/min": 0.7722375392913818, "sampling/importance_sampling_ratio/mean": 0.9995589256286621, "sampling/importance_sampling_ratio/max": 1.0348354578018188, "entropy": 0.007912849017884582, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0013699556002393365, "clip_ratio/high_max": 0.0013699556002393365, "clip_ratio/region_mean": 0.0013699556002393365, "reward_total_mean": 0.7214329242706299, "reward_meter_mean": 0.9632084965705872, "reward_meter_std": 0.07406298071146011, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.2357022613286972, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7214329242706299, "reward_total_composite_std": 0.2915029525756836, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 702.0} {"timestamp_utc": "2026-04-11T20:54:33Z", "mode": "train", "global_step": 703, "epoch": 0.027147049737411184, "loss": 0.1293, "grad_norm": 2.0049469470977783, "learning_rate": 7.872727272727273e-06, "num_tokens": 1527510.0, "completions/mean_length": 141.25, "completions/min_length": 124.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.25, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.9794438481330872, "rewards/meter/std": 0.028798731043934822, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9319754838943481, "rewards/total_composite/std": 0.10790520161390305, "reward": 0.9319754838943481, "reward_std": 0.10790519416332245, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007004235405474901, "sampling/sampling_logp_difference/max": 1.374394178390503, "sampling/importance_sampling_ratio/min": 0.25299280881881714, "sampling/importance_sampling_ratio/mean": 1.0015076398849487, "sampling/importance_sampling_ratio/max": 1.642334222793579, "entropy": 0.033183612511493266, "clip_ratio/low_mean": 0.0028273810748942196, "clip_ratio/low_min": 0.0028273810748942196, "clip_ratio/high_mean": 0.004795322893187404, "clip_ratio/high_max": 0.004795322893187404, "clip_ratio/region_mean": 0.007622703968081623, "reward_total_mean": 0.9319754838943481, "reward_meter_mean": 0.9794438481330872, "reward_meter_std": 0.028798731043934822, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9319754838943481, "reward_total_composite_std": 0.10790520161390305, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 703.0} {"timestamp_utc": "2026-04-11T20:54:38Z", "mode": "train", "global_step": 704, "epoch": 0.027185665739882608, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.86969696969697e-06, "num_tokens": 1529614.0, "completions/mean_length": 102.0, "completions/min_length": 102.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.0, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.9958475828170776, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9958475828170776, "rewards/total_composite/std": 0.0, "reward": 0.9958475828170776, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005967670003883541, "sampling/sampling_logp_difference/max": 0.09291226416826248, "sampling/importance_sampling_ratio/min": 0.9112734794616699, "sampling/importance_sampling_ratio/mean": 1.0003347396850586, "sampling/importance_sampling_ratio/max": 1.038313865661621, "entropy": 0.003974960185587406, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9958475828170776, "reward_meter_mean": 0.9958475828170776, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9958475828170776, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 704.0} {"timestamp_utc": "2026-04-11T20:54:48Z", "mode": "train", "global_step": 705, "epoch": 0.027224281742354032, "loss": -0.1759, "grad_norm": 1.0009896755218506, "learning_rate": 7.866666666666667e-06, "num_tokens": 1531473.0, "completions/mean_length": 128.375, "completions/min_length": 70.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.5714340209961, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9512093663215637, "rewards/meter/std": 0.10516034066677094, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8648210763931274, "rewards/total_composite/std": 0.3494594097137451, "reward": 0.8648210763931274, "reward_std": 0.34945937991142273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02942178212106228, "sampling/sampling_logp_difference/max": 1.039928913116455, "sampling/importance_sampling_ratio/min": 0.35347980260849, "sampling/importance_sampling_ratio/mean": 1.0000776052474976, "sampling/importance_sampling_ratio/max": 1.7531105279922485, "entropy": 0.164587477222085, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.029144452302716672, "clip_ratio/high_max": 0.029144452302716672, "clip_ratio/region_mean": 0.029144452302716672, "reward_total_mean": 0.8648210763931274, "reward_meter_mean": 0.9512093663215637, "reward_meter_std": 0.10516034066677094, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8648210763931274, "reward_total_composite_std": 0.3494594097137451, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 705.0} {"timestamp_utc": "2026-04-11T20:54:52Z", "mode": "train", "global_step": 706, "epoch": 0.027262897744825456, "loss": -0.001, "grad_norm": 0.023275254294276237, "learning_rate": 7.863636363636364e-06, "num_tokens": 1533304.0, "completions/mean_length": 63.875, "completions/min_length": 63.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9995351433753967, "rewards/meter/std": 1.702732697594911e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9995351433753967, "rewards/total_composite/std": 1.702732697594911e-05, "reward": 0.9995351433753967, "reward_std": 1.702732697594911e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0025622767861932516, "sampling/sampling_logp_difference/max": 0.2329850196838379, "sampling/importance_sampling_ratio/min": 0.7921654582023621, "sampling/importance_sampling_ratio/mean": 0.9993675351142883, "sampling/importance_sampling_ratio/max": 1.06966233253479, "entropy": 0.012379273306578398, "clip_ratio/low_mean": 0.0019841270986944437, "clip_ratio/low_min": 0.0019841270986944437, "clip_ratio/high_mean": 0.00390625, "clip_ratio/high_max": 0.00390625, "clip_ratio/region_mean": 0.005890377098694444, "reward_total_mean": 0.9995351433753967, "reward_meter_mean": 0.9995351433753967, "reward_meter_std": 1.702732697594911e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9995351433753967, "reward_total_composite_std": 1.702732697594911e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 706.0} {"timestamp_utc": "2026-04-11T20:54:58Z", "mode": "train", "global_step": 707, "epoch": 0.02730151374729688, "loss": -0.008, "grad_norm": 2.3474538326263428, "learning_rate": 7.860606060606062e-06, "num_tokens": 1535515.0, "completions/mean_length": 105.375, "completions/min_length": 101.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.375, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.9893864393234253, "rewards/meter/std": 0.00025189065490849316, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9893864393234253, "rewards/total_composite/std": 0.00025189065490849316, "reward": 0.9893864393234253, "reward_std": 0.0002518816036172211, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004941336810588837, "sampling/sampling_logp_difference/max": 0.5251731872558594, "sampling/importance_sampling_ratio/min": 0.6098865866661072, "sampling/importance_sampling_ratio/mean": 1.0015161037445068, "sampling/importance_sampling_ratio/max": 1.690751552581787, "entropy": 0.031115965684875846, "clip_ratio/low_mean": 0.002475247485563159, "clip_ratio/low_min": 0.002475247485563159, "clip_ratio/high_mean": 0.001179245300590992, "clip_ratio/high_max": 0.001179245300590992, "clip_ratio/region_mean": 0.003654492786154151, "reward_total_mean": 0.9893864393234253, "reward_meter_mean": 0.9893864393234253, "reward_meter_std": 0.00025189065490849316, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9893864393234253, "reward_total_composite_std": 0.00025189065490849316, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 707.0} {"timestamp_utc": "2026-04-11T20:55:09Z", "mode": "train", "global_step": 708, "epoch": 0.027340129749768304, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.857575757575759e-06, "num_tokens": 1537699.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9958475828170776, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.7894737124443054, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7861954569816589, "rewards/total_composite/std": 0.0, "reward": 0.7861954569816589, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7861954569816589, "reward_meter_mean": 0.9958475828170776, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.7894737124443054, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7861954569816589, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 708.0} {"timestamp_utc": "2026-04-11T20:55:14Z", "mode": "train", "global_step": 709, "epoch": 0.02737874575223973, "loss": 0.0, "grad_norm": 0.044077515602111816, "learning_rate": 7.854545454545454e-06, "num_tokens": 1539444.0, "completions/mean_length": 64.125, "completions/min_length": 64.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9995409250259399, "rewards/meter/std": 6.743495646333031e-07, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9995409250259399, "rewards/total_composite/std": 6.743495646333031e-07, "reward": 0.9995409250259399, "reward_std": 6.743495646333031e-07, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007541534956544638, "sampling/sampling_logp_difference/max": 0.7462999820709229, "sampling/importance_sampling_ratio/min": 0.47411754727363586, "sampling/importance_sampling_ratio/mean": 1.0011457204818726, "sampling/importance_sampling_ratio/max": 1.6383085250854492, "entropy": 0.034536257619038224, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005859375, "clip_ratio/high_max": 0.005859375, "clip_ratio/region_mean": 0.005859375, "reward_total_mean": 0.9995409250259399, "reward_meter_mean": 0.9995409250259399, "reward_meter_std": 6.743495646333031e-07, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9995409250259399, "reward_total_composite_std": 6.743495646333031e-07, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 709.0} {"timestamp_utc": "2026-04-11T20:55:19Z", "mode": "train", "global_step": 710, "epoch": 0.027417361754711152, "loss": 0.0192, "grad_norm": 4.497684955596924, "learning_rate": 7.851515151515152e-06, "num_tokens": 1541095.0, "completions/mean_length": 56.375, "completions/min_length": 56.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9946244359016418, "rewards/meter/std": 0.00035911225131712854, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9946244359016418, "rewards/total_composite/std": 0.00035911225131712854, "reward": 0.9946244359016418, "reward_std": 0.0003591080312617123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008529079146683216, "sampling/sampling_logp_difference/max": 0.4691757559776306, "sampling/importance_sampling_ratio/min": 0.6307634711265564, "sampling/importance_sampling_ratio/mean": 1.0022271871566772, "sampling/importance_sampling_ratio/max": 1.5986759662628174, "entropy": 0.06256510340608656, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/high_mean": 0.008928571827709675, "clip_ratio/high_max": 0.008928571827709675, "clip_ratio/region_mean": 0.01316585997119546, "reward_total_mean": 0.9946244359016418, "reward_meter_mean": 0.9946244359016418, "reward_meter_std": 0.00035911225131712854, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9946244359016418, "reward_total_composite_std": 0.00035911225131712854, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 710.0} {"timestamp_utc": "2026-04-11T20:55:25Z", "mode": "train", "global_step": 711, "epoch": 0.027455977757182577, "loss": 0.0975, "grad_norm": 3.473660707473755, "learning_rate": 7.848484848484849e-06, "num_tokens": 1543381.0, "completions/mean_length": 109.75, "completions/min_length": 102.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.75, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.9711757898330688, "rewards/meter/std": 0.051759473979473114, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9711757898330688, "rewards/total_composite/std": 0.051759473979473114, "reward": 0.9711757898330688, "reward_std": 0.05175946280360222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011688003316521645, "sampling/sampling_logp_difference/max": 2.8346152305603027, "sampling/importance_sampling_ratio/min": 0.05874112248420715, "sampling/importance_sampling_ratio/mean": 0.9987711310386658, "sampling/importance_sampling_ratio/max": 1.8414362668991089, "entropy": 0.031717493664473295, "clip_ratio/low_mean": 0.002659574383869767, "clip_ratio/low_min": 0.002659574383869767, "clip_ratio/high_mean": 0.0024509804788976908, "clip_ratio/high_max": 0.0024509804788976908, "clip_ratio/region_mean": 0.005110554862767458, "reward_total_mean": 0.9711757898330688, "reward_meter_mean": 0.9711757898330688, "reward_meter_std": 0.051759473979473114, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9711757898330688, "reward_total_composite_std": 0.051759473979473114, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 711.0} {"timestamp_utc": "2026-04-11T20:55:29Z", "mode": "train", "global_step": 712, "epoch": 0.027494593759654, "loss": 0.0067, "grad_norm": 4.340436935424805, "learning_rate": 7.845454545454546e-06, "num_tokens": 1544835.0, "completions/mean_length": 34.75, "completions/min_length": 33.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9944402575492859, "rewards/meter/std": 0.00012188738764962181, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9944402575492859, "rewards/total_composite/std": 0.00012188738764962181, "reward": 0.9944402575492859, "reward_std": 0.00012190965935587883, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014197859913110733, "sampling/sampling_logp_difference/max": 1.7028403282165527, "sampling/importance_sampling_ratio/min": 0.18216538429260254, "sampling/importance_sampling_ratio/mean": 0.9965629577636719, "sampling/importance_sampling_ratio/max": 1.229400634765625, "entropy": 0.0445503736846149, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/region_mean": 0.01093073608353734, "reward_total_mean": 0.9944402575492859, "reward_meter_mean": 0.9944402575492859, "reward_meter_std": 0.00012188738764962181, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9944402575492859, "reward_total_composite_std": 0.00012188738764962181, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 712.0} {"timestamp_utc": "2026-04-11T20:55:34Z", "mode": "train", "global_step": 713, "epoch": 0.027533209762125425, "loss": 0.0743, "grad_norm": 7.83790397644043, "learning_rate": 7.842424242424243e-06, "num_tokens": 1546527.0, "completions/mean_length": 62.5, "completions/min_length": 52.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.21346108615398407, "rewards/meter/std": 0.3139375150203705, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.17752352356910706, "rewards/total_composite/std": 0.3206353485584259, "reward": 0.17752352356910706, "reward_std": 0.3206353187561035, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027129970490932465, "sampling/sampling_logp_difference/max": 1.384993314743042, "sampling/importance_sampling_ratio/min": 0.250325471162796, "sampling/importance_sampling_ratio/mean": 1.0073413848876953, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14027372561395168, "clip_ratio/low_mean": 0.0156496997224167, "clip_ratio/low_min": 0.0156496997224167, "clip_ratio/high_mean": 0.011568509973585606, "clip_ratio/high_max": 0.011568509973585606, "clip_ratio/region_mean": 0.027218209696002305, "reward_total_mean": 0.17752352356910706, "reward_meter_mean": 0.21346108615398407, "reward_meter_std": 0.3139375150203705, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.17752352356910706, "reward_total_composite_std": 0.3206353485584259, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 713.0} {"timestamp_utc": "2026-04-11T20:55:39Z", "mode": "train", "global_step": 714, "epoch": 0.02757182576459685, "loss": 0.0041, "grad_norm": 2.5651803016662598, "learning_rate": 7.83939393939394e-06, "num_tokens": 1548387.0, "completions/mean_length": 79.5, "completions/min_length": 76.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.5, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9946976900100708, "rewards/meter/std": 0.006162641569972038, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9946976900100708, "rewards/total_composite/std": 0.006162641569972038, "reward": 0.9946976900100708, "reward_std": 0.006162638776004314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010636840015649796, "sampling/sampling_logp_difference/max": 1.224252700805664, "sampling/importance_sampling_ratio/min": 0.2939773201942444, "sampling/importance_sampling_ratio/mean": 1.0002338886260986, "sampling/importance_sampling_ratio/max": 1.345471978187561, "entropy": 0.05185921536758542, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0015625000232830644, "clip_ratio/high_max": 0.0015625000232830644, "clip_ratio/region_mean": 0.0015625000232830644, "reward_total_mean": 0.9946976900100708, "reward_meter_mean": 0.9946976900100708, "reward_meter_std": 0.006162641569972038, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9946976900100708, "reward_total_composite_std": 0.006162641569972038, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 714.0} {"timestamp_utc": "2026-04-11T20:55:44Z", "mode": "train", "global_step": 715, "epoch": 0.027610441767068273, "loss": 0.0088, "grad_norm": 2.420339584350586, "learning_rate": 7.836363636363638e-06, "num_tokens": 1550163.0, "completions/mean_length": 64.0, "completions/min_length": 63.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.999401330947876, "rewards/meter/std": 0.00035342175397090614, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.999401330947876, "rewards/total_composite/std": 0.00035342175397090614, "reward": 0.999401330947876, "reward_std": 0.0003534228599164635, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009754814207553864, "sampling/sampling_logp_difference/max": 1.2119450569152832, "sampling/importance_sampling_ratio/min": 0.2976178228855133, "sampling/importance_sampling_ratio/mean": 0.9978449940681458, "sampling/importance_sampling_ratio/max": 1.4095262289047241, "entropy": 0.032442330149933696, "clip_ratio/low_mean": 0.003968254197388887, "clip_ratio/low_min": 0.003968254197388887, "clip_ratio/high_mean": 0.009765625, "clip_ratio/high_max": 0.009765625, "clip_ratio/region_mean": 0.013733879197388887, "reward_total_mean": 0.999401330947876, "reward_meter_mean": 0.999401330947876, "reward_meter_std": 0.00035342175397090614, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.999401330947876, "reward_total_composite_std": 0.00035342175397090614, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 715.0} {"timestamp_utc": "2026-04-11T20:55:54Z", "mode": "train", "global_step": 716, "epoch": 0.027649057769539697, "loss": -0.1529, "grad_norm": 0.262566477060318, "learning_rate": 7.833333333333333e-06, "num_tokens": 1551753.0, "completions/mean_length": 112.75, "completions/min_length": 54.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 55.71428680419922, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.8704155683517456, "rewards/meter/std": 0.351701021194458, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8704155683517456, "rewards/total_composite/std": 0.351701021194458, "reward": 0.8704155683517456, "reward_std": 0.351701021194458, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005107759032398462, "sampling/sampling_logp_difference/max": 0.5069513320922852, "sampling/importance_sampling_ratio/min": 0.6023290753364563, "sampling/importance_sampling_ratio/mean": 1.0019290447235107, "sampling/importance_sampling_ratio/max": 1.2365697622299194, "entropy": 0.040907534305006266, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.002314814832061529, "clip_ratio/high_max": 0.002314814832061529, "clip_ratio/region_mean": 0.002314814832061529, "reward_total_mean": 0.8704155683517456, "reward_meter_mean": 0.8704155683517456, "reward_meter_std": 0.351701021194458, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8704155683517456, "reward_total_composite_std": 0.351701021194458, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 716.0} {"timestamp_utc": "2026-04-11T20:55:59Z", "mode": "train", "global_step": 717, "epoch": 0.02768767377201112, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.83030303030303e-06, "num_tokens": 1553585.0, "completions/mean_length": 72.0, "completions/min_length": 72.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9891483783721924, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9891483783721924, "rewards/total_composite/std": 0.0, "reward": 0.9891483783721924, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0008889577584341168, "sampling/sampling_logp_difference/max": 0.019037650898098946, "sampling/importance_sampling_ratio/min": 0.994313657283783, "sampling/importance_sampling_ratio/mean": 1.0008010864257812, "sampling/importance_sampling_ratio/max": 1.019219994544983, "entropy": 0.009976035333238542, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9891483783721924, "reward_meter_mean": 0.9891483783721924, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9891483783721924, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 717.0} {"timestamp_utc": "2026-04-11T20:56:03Z", "mode": "train", "global_step": 718, "epoch": 0.027726289774482545, "loss": -0.0099, "grad_norm": 4.412846565246582, "learning_rate": 7.827272727272728e-06, "num_tokens": 1555360.0, "completions/mean_length": 49.875, "completions/min_length": 48.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.875, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8061593770980835, "rewards/meter/std": 0.2929365634918213, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8061593770980835, "rewards/total_composite/std": 0.2929365634918213, "reward": 0.8061593770980835, "reward_std": 0.2929365336894989, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036188237369060516, "sampling/sampling_logp_difference/max": 1.053207516670227, "sampling/importance_sampling_ratio/min": 0.3488171100616455, "sampling/importance_sampling_ratio/mean": 0.9926144480705261, "sampling/importance_sampling_ratio/max": 1.436261534690857, "entropy": 0.17880077846348286, "clip_ratio/low_mean": 0.007604166632518172, "clip_ratio/low_min": 0.007604166632518172, "clip_ratio/high_mean": 0.027567277662456036, "clip_ratio/high_max": 0.027567277662456036, "clip_ratio/region_mean": 0.03517144429497421, "reward_total_mean": 0.8061593770980835, "reward_meter_mean": 0.8061593770980835, "reward_meter_std": 0.2929365634918213, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8061593770980835, "reward_total_composite_std": 0.2929365634918213, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 718.0} {"timestamp_utc": "2026-04-11T20:56:08Z", "mode": "train", "global_step": 719, "epoch": 0.02776490577695397, "loss": 0.0194, "grad_norm": 0.5022082924842834, "learning_rate": 7.824242424242425e-06, "num_tokens": 1557125.0, "completions/mean_length": 53.625, "completions/min_length": 51.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9899463653564453, "rewards/meter/std": 0.0006506209028884768, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9899463653564453, "rewards/total_composite/std": 0.0006506209028884768, "reward": 0.9899463653564453, "reward_std": 0.0006506269564852118, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008551125414669514, "sampling/sampling_logp_difference/max": 2.286591053009033, "sampling/importance_sampling_ratio/min": 0.10161226242780685, "sampling/importance_sampling_ratio/mean": 0.9990531206130981, "sampling/importance_sampling_ratio/max": 1.1061651706695557, "entropy": 0.019666927866637707, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0024509804788976908, "clip_ratio/high_max": 0.0024509804788976908, "clip_ratio/region_mean": 0.0024509804788976908, "reward_total_mean": 0.9899463653564453, "reward_meter_mean": 0.9899463653564453, "reward_meter_std": 0.0006506209028884768, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9899463653564453, "reward_total_composite_std": 0.0006506209028884768, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 719.0} {"timestamp_utc": "2026-04-11T20:56:13Z", "mode": "train", "global_step": 720, "epoch": 0.027803521779425393, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.821212121212122e-06, "num_tokens": 1558947.0, "completions/mean_length": 50.75, "completions/min_length": 50.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9915565848350525, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9915565848350525, "rewards/total_composite/std": 0.0, "reward": 0.9915565848350525, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.01000992488116026, "sampling/sampling_logp_difference/max": 0.9073116779327393, "sampling/importance_sampling_ratio/min": 0.6772187352180481, "sampling/importance_sampling_ratio/mean": 1.0001487731933594, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03703967132605612, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9915565848350525, "reward_meter_mean": 0.9915565848350525, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9915565848350525, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 720.0} {"timestamp_utc": "2026-04-11T20:56:18Z", "mode": "train", "global_step": 721, "epoch": 0.027842137781896818, "loss": -0.0181, "grad_norm": 3.8514387607574463, "learning_rate": 7.81818181818182e-06, "num_tokens": 1560642.0, "completions/mean_length": 60.875, "completions/min_length": 57.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9944725036621094, "rewards/meter/std": 0.002843559952452779, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9944725036621094, "rewards/total_composite/std": 0.002843559952452779, "reward": 0.9944725036621094, "reward_std": 0.0028435522690415382, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029106715694069862, "sampling/sampling_logp_difference/max": 1.0956978797912598, "sampling/importance_sampling_ratio/min": 0.3343062102794647, "sampling/importance_sampling_ratio/mean": 0.9935939311981201, "sampling/importance_sampling_ratio/max": 1.666185975074768, "entropy": 0.11305387504398823, "clip_ratio/low_mean": 0.0129739074036479, "clip_ratio/low_min": 0.0129739074036479, "clip_ratio/high_mean": 0.019668605644255877, "clip_ratio/high_max": 0.019668605644255877, "clip_ratio/region_mean": 0.032642513047903776, "reward_total_mean": 0.9944725036621094, "reward_meter_mean": 0.9944725036621094, "reward_meter_std": 0.002843559952452779, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9944725036621094, "reward_total_composite_std": 0.002843559952452779, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 721.0} {"timestamp_utc": "2026-04-11T20:56:24Z", "mode": "train", "global_step": 722, "epoch": 0.02788075378436824, "loss": 0.0433, "grad_norm": 2.7389070987701416, "learning_rate": 7.815151515151515e-06, "num_tokens": 1563363.0, "completions/mean_length": 153.125, "completions/min_length": 141.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 153.125, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.995931088924408, "rewards/meter/std": 0.002105413004755974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.995931088924408, "rewards/total_composite/std": 0.002105413004755974, "reward": 0.995931088924408, "reward_std": 0.002105406019836664, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012107064947485924, "sampling/sampling_logp_difference/max": 0.8693251609802246, "sampling/importance_sampling_ratio/min": 0.41923436522483826, "sampling/importance_sampling_ratio/mean": 1.0028266906738281, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04748309962451458, "clip_ratio/low_mean": 0.005510725372005254, "clip_ratio/low_min": 0.005510725372005254, "clip_ratio/high_mean": 0.003403303329832852, "clip_ratio/high_max": 0.003403303329832852, "clip_ratio/region_mean": 0.008914028701838106, "reward_total_mean": 0.995931088924408, "reward_meter_mean": 0.995931088924408, "reward_meter_std": 0.002105413004755974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.995931088924408, "reward_total_composite_std": 0.002105413004755974, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 722.0} {"timestamp_utc": "2026-04-11T20:56:32Z", "mode": "train", "global_step": 723, "epoch": 0.027919369786839666, "loss": 0.0193, "grad_norm": 1.9227861166000366, "learning_rate": 7.812121212121213e-06, "num_tokens": 1567497.0, "completions/mean_length": 316.75, "completions/min_length": 303.0, "completions/max_length": 334.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 316.75, "completions/min_terminated_length": 303.0, "completions/max_terminated_length": 334.0, "rewards/meter/mean": 0.9993918538093567, "rewards/meter/std": 4.8108842747751623e-05, "rewards/count_adherence/mean": 0.734375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7339285016059875, "rewards/total_composite/std": 0.044169094413518906, "reward": 0.7339285016059875, "reward_std": 0.0441691055893898, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003939645830541849, "sampling/sampling_logp_difference/max": 1.0573415756225586, "sampling/importance_sampling_ratio/min": 0.5264492034912109, "sampling/importance_sampling_ratio/mean": 1.0008435249328613, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02318198117427528, "clip_ratio/low_mean": 0.0007485030218958855, "clip_ratio/low_min": 0.0007485030218958855, "clip_ratio/high_mean": 0.0035521415702532977, "clip_ratio/high_max": 0.0035521415702532977, "clip_ratio/region_mean": 0.004300644592149183, "reward_total_mean": 0.7339285016059875, "reward_meter_mean": 0.9993918538093567, "reward_meter_std": 4.8108842747751623e-05, "reward_count_adherence_mean": 0.734375, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7339285016059875, "reward_total_composite_std": 0.044169094413518906, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 723.0} {"timestamp_utc": "2026-04-11T20:56:36Z", "mode": "train", "global_step": 724, "epoch": 0.02795798578931109, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.80909090909091e-06, "num_tokens": 1568993.0, "completions/mean_length": 34.0, "completions/min_length": 34.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9958475828170776, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9958475828170776, "rewards/total_composite/std": 0.0, "reward": 0.9958475828170776, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.002870172029361129, "sampling/sampling_logp_difference/max": 0.10669367015361786, "sampling/importance_sampling_ratio/min": 0.9087722897529602, "sampling/importance_sampling_ratio/mean": 1.001280665397644, "sampling/importance_sampling_ratio/max": 1.112593412399292, "entropy": 0.022150606149807572, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9958475828170776, "reward_meter_mean": 0.9958475828170776, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9958475828170776, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 724.0} {"timestamp_utc": "2026-04-11T20:56:46Z", "mode": "train", "global_step": 725, "epoch": 0.027996601791782514, "loss": -0.1233, "grad_norm": 0.8104729652404785, "learning_rate": 7.806060606060607e-06, "num_tokens": 1570534.0, "completions/mean_length": 231.625, "completions/min_length": 62.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 63.400001525878906, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.866460919380188, "rewards/meter/std": 0.3406028747558594, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.6204345226287842, "rewards/total_composite/std": 0.5137822031974792, "reward": 0.6204345226287842, "reward_std": 0.5137822031974792, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.037449613213539124, "sampling/sampling_logp_difference/max": 0.7609826326370239, "sampling/importance_sampling_ratio/min": 0.4672071039676666, "sampling/importance_sampling_ratio/mean": 1.010333776473999, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16976340487599373, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.01953897369094193, "clip_ratio/high_max": 0.01953897369094193, "clip_ratio/region_mean": 0.01953897369094193, "reward_total_mean": 0.6204345226287842, "reward_meter_mean": 0.866460919380188, "reward_meter_std": 0.3406028747558594, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.6204345226287842, "reward_total_composite_std": 0.5137822031974792, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 725.0} {"timestamp_utc": "2026-04-11T20:56:51Z", "mode": "train", "global_step": 726, "epoch": 0.028035217794253938, "loss": 0.036, "grad_norm": 2.919178009033203, "learning_rate": 7.803030303030303e-06, "num_tokens": 1572382.0, "completions/mean_length": 80.0, "completions/min_length": 77.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.9245812892913818, "rewards/meter/std": 0.2059229463338852, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9245812892913818, "rewards/total_composite/std": 0.2059229463338852, "reward": 0.9245812892913818, "reward_std": 0.2059229463338852, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01003341656178236, "sampling/sampling_logp_difference/max": 0.9528336524963379, "sampling/importance_sampling_ratio/min": 0.38564667105674744, "sampling/importance_sampling_ratio/mean": 1.0042144060134888, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.056009670021012425, "clip_ratio/low_mean": 0.0029411765281111, "clip_ratio/low_min": 0.0029411765281111, "clip_ratio/high_mean": 0.004807692486792803, "clip_ratio/high_max": 0.004807692486792803, "clip_ratio/region_mean": 0.007748869014903903, "reward_total_mean": 0.9245812892913818, "reward_meter_mean": 0.9245812892913818, "reward_meter_std": 0.2059229463338852, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9245812892913818, "reward_total_composite_std": 0.2059229463338852, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 726.0} {"timestamp_utc": "2026-04-11T20:56:56Z", "mode": "train", "global_step": 727, "epoch": 0.028073833796725362, "loss": 0.0033, "grad_norm": 2.438025951385498, "learning_rate": 7.800000000000002e-06, "num_tokens": 1574303.0, "completions/mean_length": 62.125, "completions/min_length": 62.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.999412477016449, "rewards/meter/std": 6.900283915456384e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.999412477016449, "rewards/total_composite/std": 6.900283915456384e-05, "reward": 0.999412477016449, "reward_std": 6.900283915456384e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025097455829381943, "sampling/sampling_logp_difference/max": 0.9476951360702515, "sampling/importance_sampling_ratio/min": 0.3956271708011627, "sampling/importance_sampling_ratio/mean": 1.0035431385040283, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12215746659785509, "clip_ratio/low_mean": 0.012000767979770899, "clip_ratio/low_min": 0.012000767979770899, "clip_ratio/high_mean": 0.012096773833036423, "clip_ratio/high_max": 0.012096773833036423, "clip_ratio/region_mean": 0.02409754181280732, "reward_total_mean": 0.999412477016449, "reward_meter_mean": 0.999412477016449, "reward_meter_std": 6.900283915456384e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.999412477016449, "reward_total_composite_std": 6.900283915456384e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 727.0} {"timestamp_utc": "2026-04-11T20:57:07Z", "mode": "train", "global_step": 728, "epoch": 0.028112449799196786, "loss": -0.2244, "grad_norm": 0.9194737076759338, "learning_rate": 7.796969696969697e-06, "num_tokens": 1576765.0, "completions/mean_length": 167.75, "completions/min_length": 91.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 118.5714340209961, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.8854305744171143, "rewards/meter/std": 0.32155510783195496, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8430001735687256, "rewards/total_composite/std": 0.3516468405723572, "reward": 0.8430001735687256, "reward_std": 0.3516468107700348, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029425477609038353, "sampling/sampling_logp_difference/max": 6.463409423828125, "sampling/importance_sampling_ratio/min": 0.0015594697324559093, "sampling/importance_sampling_ratio/mean": 1.00081467628479, "sampling/importance_sampling_ratio/max": 1.97501802444458, "entropy": 0.07004146743565798, "clip_ratio/low_mean": 0.004120879340916872, "clip_ratio/low_min": 0.004120879340916872, "clip_ratio/high_mean": 0.009138820576481521, "clip_ratio/high_max": 0.009138820576481521, "clip_ratio/region_mean": 0.013259699917398393, "reward_total_mean": 0.8430001735687256, "reward_meter_mean": 0.8854305744171143, "reward_meter_std": 0.32155510783195496, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.18600596487522125, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8430001735687256, "reward_total_composite_std": 0.3516468405723572, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 728.0} {"timestamp_utc": "2026-04-11T20:57:12Z", "mode": "train", "global_step": 729, "epoch": 0.02815106580166821, "loss": 0.0025, "grad_norm": 1.2327404022216797, "learning_rate": 7.793939393939394e-06, "num_tokens": 1578763.0, "completions/mean_length": 74.75, "completions/min_length": 74.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.75, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9901421666145325, "rewards/meter/std": 0.0032967478036880493, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9901421666145325, "rewards/total_composite/std": 0.0032967478036880493, "reward": 0.9901421666145325, "reward_std": 0.003296738490462303, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005285956896841526, "sampling/sampling_logp_difference/max": 0.7588220834732056, "sampling/importance_sampling_ratio/min": 0.902445375919342, "sampling/importance_sampling_ratio/mean": 1.004712462425232, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.023266208823770285, "clip_ratio/low_mean": 0.0016666667070239782, "clip_ratio/low_min": 0.0016666667070239782, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0016666667070239782, "reward_total_mean": 0.9901421666145325, "reward_meter_mean": 0.9901421666145325, "reward_meter_std": 0.0032967478036880493, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9901421666145325, "reward_total_composite_std": 0.0032967478036880493, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 729.0} {"timestamp_utc": "2026-04-11T20:57:17Z", "mode": "train", "global_step": 730, "epoch": 0.028189681804139635, "loss": 0.0151, "grad_norm": 3.6636853218078613, "learning_rate": 7.790909090909092e-06, "num_tokens": 1580990.0, "completions/mean_length": 116.375, "completions/min_length": 115.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.375, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9987906217575073, "rewards/meter/std": 0.0006424913299269974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9987906217575073, "rewards/total_composite/std": 0.0006424913299269974, "reward": 0.9987906217575073, "reward_std": 0.0006424955790862441, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003841625526547432, "sampling/sampling_logp_difference/max": 0.344424843788147, "sampling/importance_sampling_ratio/min": 0.7086277604103088, "sampling/importance_sampling_ratio/mean": 1.0024141073226929, "sampling/importance_sampling_ratio/max": 1.295914649963379, "entropy": 0.022385249380022287, "clip_ratio/low_mean": 0.001033057807944715, "clip_ratio/low_min": 0.001033057807944715, "clip_ratio/high_mean": 0.0010869564721360803, "clip_ratio/high_max": 0.0010869564721360803, "clip_ratio/region_mean": 0.0021200142800807953, "reward_total_mean": 0.9987906217575073, "reward_meter_mean": 0.9987906217575073, "reward_meter_std": 0.0006424913299269974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9987906217575073, "reward_total_composite_std": 0.0006424913299269974, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 730.0} {"timestamp_utc": "2026-04-11T20:57:21Z", "mode": "train", "global_step": 731, "epoch": 0.02822829780661106, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.787878787878789e-06, "num_tokens": 1582422.0, "completions/mean_length": 28.0, "completions/min_length": 28.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9950272440910339, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9950272440910339, "rewards/total_composite/std": 0.0, "reward": 0.9950272440910339, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0016047912649810314, "sampling/sampling_logp_difference/max": 0.027983784675598145, "sampling/importance_sampling_ratio/min": 0.9988099932670593, "sampling/importance_sampling_ratio/mean": 1.0015828609466553, "sampling/importance_sampling_ratio/max": 1.028378963470459, "entropy": 0.012361339526250958, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9950272440910339, "reward_meter_mean": 0.9950272440910339, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9950272440910339, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 731.0} {"timestamp_utc": "2026-04-11T20:57:31Z", "mode": "train", "global_step": 732, "epoch": 0.028266913809082483, "loss": -0.1665, "grad_norm": 0.6903420686721802, "learning_rate": 7.784848484848484e-06, "num_tokens": 1584201.0, "completions/mean_length": 121.375, "completions/min_length": 65.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 65.5714340209961, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9502468109130859, "rewards/meter/std": 0.13813894987106323, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8742003440856934, "rewards/total_composite/std": 0.3532305657863617, "reward": 0.8742003440856934, "reward_std": 0.3532305359840393, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013370465487241745, "sampling/sampling_logp_difference/max": 1.5705301761627197, "sampling/importance_sampling_ratio/min": 0.2079349011182785, "sampling/importance_sampling_ratio/mean": 1.0019134283065796, "sampling/importance_sampling_ratio/max": 1.8689512014389038, "entropy": 0.07014387752860785, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007575757801532745, "clip_ratio/high_max": 0.007575757801532745, "clip_ratio/region_mean": 0.007575757801532745, "reward_total_mean": 0.8742003440856934, "reward_meter_mean": 0.9502468109130859, "reward_meter_std": 0.13813894987106323, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8742003440856934, "reward_total_composite_std": 0.3532305657863617, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 732.0} {"timestamp_utc": "2026-04-11T20:57:38Z", "mode": "train", "global_step": 733, "epoch": 0.028305529811553907, "loss": -0.0281, "grad_norm": 1.2909003496170044, "learning_rate": 7.781818181818183e-06, "num_tokens": 1588025.0, "completions/mean_length": 258.0, "completions/min_length": 247.0, "completions/max_length": 267.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 258.0, "completions/min_terminated_length": 247.0, "completions/max_terminated_length": 267.0, "rewards/meter/mean": 0.9988299608230591, "rewards/meter/std": 0.0010553781175985932, "rewards/count_adherence/mean": 0.9464285969734192, "rewards/count_adherence/std": 0.07393559068441391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9453060626983643, "rewards/total_composite/std": 0.07364311069250107, "reward": 0.9453060626983643, "reward_std": 0.07364311069250107, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005114637780934572, "sampling/sampling_logp_difference/max": 1.0911688804626465, "sampling/importance_sampling_ratio/min": 0.3358237147331238, "sampling/importance_sampling_ratio/mean": 0.9997689723968506, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.020183743443340063, "clip_ratio/low_mean": 0.0020121458219364285, "clip_ratio/low_min": 0.0020121458219364285, "clip_ratio/high_mean": 0.0034164884127676487, "clip_ratio/high_max": 0.0034164884127676487, "clip_ratio/region_mean": 0.005428634234704077, "reward_total_mean": 0.9453060626983643, "reward_meter_mean": 0.9988299608230591, "reward_meter_std": 0.0010553781175985932, "reward_count_adherence_mean": 0.9464285969734192, "reward_count_adherence_std": 0.07393559068441391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9453060626983643, "reward_total_composite_std": 0.07364311069250107, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 733.0} {"timestamp_utc": "2026-04-11T20:57:44Z", "mode": "train", "global_step": 734, "epoch": 0.02834414581402533, "loss": -0.0001, "grad_norm": 0.2421007603406906, "learning_rate": 7.778787878787879e-06, "num_tokens": 1590169.0, "completions/mean_length": 114.0, "completions/min_length": 114.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.0, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9950557947158813, "rewards/meter/std": 4.482320946408436e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9950557947158813, "rewards/total_composite/std": 4.482320946408436e-05, "reward": 0.9950557947158813, "reward_std": 4.4808126403950155e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0014124212320894003, "sampling/sampling_logp_difference/max": 0.3996305465698242, "sampling/importance_sampling_ratio/min": 0.6705677509307861, "sampling/importance_sampling_ratio/mean": 1.0002965927124023, "sampling/importance_sampling_ratio/max": 1.0817526578903198, "entropy": 0.010225348232779652, "clip_ratio/low_mean": 0.0010964912362396717, "clip_ratio/low_min": 0.0010964912362396717, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0010964912362396717, "reward_total_mean": 0.9950557947158813, "reward_meter_mean": 0.9950557947158813, "reward_meter_std": 4.482320946408436e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9950557947158813, "reward_total_composite_std": 4.482320946408436e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 734.0} {"timestamp_utc": "2026-04-11T20:57:49Z", "mode": "train", "global_step": 735, "epoch": 0.028382761816496755, "loss": 0.0021, "grad_norm": 1.915930151939392, "learning_rate": 7.775757575757576e-06, "num_tokens": 1592457.0, "completions/mean_length": 108.0, "completions/min_length": 108.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.0, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9895275235176086, "rewards/meter/std": 0.00011579846614040434, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9895275235176086, "rewards/total_composite/std": 0.00011579846614040434, "reward": 0.9895275235176086, "reward_std": 0.00011580749560380355, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005119105335325003, "sampling/sampling_logp_difference/max": 0.9385128021240234, "sampling/importance_sampling_ratio/min": 0.3912091851234436, "sampling/importance_sampling_ratio/mean": 1.0001648664474487, "sampling/importance_sampling_ratio/max": 1.1975129842758179, "entropy": 0.02697245148010552, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.002314814832061529, "clip_ratio/high_max": 0.002314814832061529, "clip_ratio/region_mean": 0.002314814832061529, "reward_total_mean": 0.9895275235176086, "reward_meter_mean": 0.9895275235176086, "reward_meter_std": 0.00011579846614040434, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9895275235176086, "reward_total_composite_std": 0.00011579846614040434, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 735.0} {"timestamp_utc": "2026-04-11T20:57:54Z", "mode": "train", "global_step": 736, "epoch": 0.02842137781896818, "loss": -0.0191, "grad_norm": 5.000483989715576, "learning_rate": 7.772727272727273e-06, "num_tokens": 1594229.0, "completions/mean_length": 57.5, "completions/min_length": 57.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9954984188079834, "rewards/meter/std": 0.0011250048410147429, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9954984188079834, "rewards/total_composite/std": 0.0011250048410147429, "reward": 0.9954984188079834, "reward_std": 0.0011250197421759367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005961274262517691, "sampling/sampling_logp_difference/max": 0.4364249110221863, "sampling/importance_sampling_ratio/min": 0.6463430523872375, "sampling/importance_sampling_ratio/mean": 1.000386118888855, "sampling/importance_sampling_ratio/max": 1.105066180229187, "entropy": 0.02583174896426499, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9954984188079834, "reward_meter_mean": 0.9954984188079834, "reward_meter_std": 0.0011250048410147429, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9954984188079834, "reward_total_composite_std": 0.0011250048410147429, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 736.0} {"timestamp_utc": "2026-04-11T20:57:59Z", "mode": "train", "global_step": 737, "epoch": 0.028459993821439603, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.76969696969697e-06, "num_tokens": 1596192.0, "completions/mean_length": 70.375, "completions/min_length": 70.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9913077354431152, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9913077354431152, "rewards/total_composite/std": 0.0, "reward": 0.9913077354431152, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00430458365008235, "sampling/sampling_logp_difference/max": 0.7894887924194336, "sampling/importance_sampling_ratio/min": 0.4540768563747406, "sampling/importance_sampling_ratio/mean": 0.9998700618743896, "sampling/importance_sampling_ratio/max": 1.3126484155654907, "entropy": 0.017793453531339765, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9913077354431152, "reward_meter_mean": 0.9913077354431152, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9913077354431152, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 737.0} {"timestamp_utc": "2026-04-11T20:58:04Z", "mode": "train", "global_step": 738, "epoch": 0.028498609823911027, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.766666666666666e-06, "num_tokens": 1597800.0, "completions/mean_length": 48.0, "completions/min_length": 48.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9915565848350525, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9915565848350525, "rewards/total_composite/std": 0.0, "reward": 0.9915565848350525, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.011907785199582577, "sampling/sampling_logp_difference/max": 3.073211908340454, "sampling/importance_sampling_ratio/min": 0.046272292733192444, "sampling/importance_sampling_ratio/mean": 0.9955384135246277, "sampling/importance_sampling_ratio/max": 1.0766491889953613, "entropy": 0.013242437271401286, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9915565848350525, "reward_meter_mean": 0.9915565848350525, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9915565848350525, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 738.0} {"timestamp_utc": "2026-04-11T20:58:14Z", "mode": "train", "global_step": 739, "epoch": 0.02853722582638245, "loss": -0.2454, "grad_norm": 0.9947829246520996, "learning_rate": 7.763636363636364e-06, "num_tokens": 1600029.0, "completions/mean_length": 231.625, "completions/min_length": 122.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 138.1666717529297, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.8841496706008911, "rewards/meter/std": 0.3131726086139679, "rewards/count_adherence/mean": 0.824999988079071, "rewards/count_adherence/std": 0.19820624589920044, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.6715205907821655, "rewards/total_composite/std": 0.42465120553970337, "reward": 0.6715205907821655, "reward_std": 0.42465120553970337, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009153744205832481, "sampling/sampling_logp_difference/max": 1.2841796875, "sampling/importance_sampling_ratio/min": 0.27687761187553406, "sampling/importance_sampling_ratio/mean": 0.9991498589515686, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.030928871943615377, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004460687632672489, "clip_ratio/high_max": 0.004460687632672489, "clip_ratio/region_mean": 0.004460687632672489, "reward_total_mean": 0.6715205907821655, "reward_meter_mean": 0.8841496706008911, "reward_meter_std": 0.3131726086139679, "reward_count_adherence_mean": 0.824999988079071, "reward_count_adherence_std": 0.19820624589920044, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.6715205907821655, "reward_total_composite_std": 0.42465120553970337, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 739.0} {"timestamp_utc": "2026-04-11T20:58:20Z", "mode": "train", "global_step": 740, "epoch": 0.028575841828853876, "loss": 0.0207, "grad_norm": 2.150345802307129, "learning_rate": 7.76060606060606e-06, "num_tokens": 1602719.0, "completions/mean_length": 154.25, "completions/min_length": 131.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.25, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.9964789152145386, "rewards/meter/std": 0.0011265217326581478, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9653591513633728, "rewards/total_composite/std": 0.0882880762219429, "reward": 0.9653591513633728, "reward_std": 0.0882880836725235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012078757397830486, "sampling/sampling_logp_difference/max": 1.4978318214416504, "sampling/importance_sampling_ratio/min": 0.22361448407173157, "sampling/importance_sampling_ratio/mean": 1.0022939443588257, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06363651528954506, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0068271911004558206, "clip_ratio/high_max": 0.0068271911004558206, "clip_ratio/region_mean": 0.0068271911004558206, "reward_total_mean": 0.9653591513633728, "reward_meter_mean": 0.9964789152145386, "reward_meter_std": 0.0011265217326581478, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9653591513633728, "reward_total_composite_std": 0.0882880762219429, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 740.0} {"timestamp_utc": "2026-04-11T20:58:25Z", "mode": "train", "global_step": 741, "epoch": 0.0286144578313253, "loss": 0.0229, "grad_norm": 8.573525428771973, "learning_rate": 7.757575757575758e-06, "num_tokens": 1604615.0, "completions/mean_length": 64.0, "completions/min_length": 61.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9985337853431702, "rewards/meter/std": 0.002053620293736458, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985337853431702, "rewards/total_composite/std": 0.002053620293736458, "reward": 0.9985337853431702, "reward_std": 0.002053626114502549, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03388599306344986, "sampling/sampling_logp_difference/max": 1.2888221740722656, "sampling/importance_sampling_ratio/min": 0.27559518814086914, "sampling/importance_sampling_ratio/mean": 1.0104244947433472, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16866595903411508, "clip_ratio/low_mean": 0.009328357875347137, "clip_ratio/low_min": 0.009328357875347137, "clip_ratio/high_mean": 0.015551135409623384, "clip_ratio/high_max": 0.015551135409623384, "clip_ratio/region_mean": 0.024879493284970522, "reward_total_mean": 0.9985337853431702, "reward_meter_mean": 0.9985337853431702, "reward_meter_std": 0.002053620293736458, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985337853431702, "reward_total_composite_std": 0.002053620293736458, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 741.0} {"timestamp_utc": "2026-04-11T20:58:29Z", "mode": "train", "global_step": 742, "epoch": 0.028653073833796724, "loss": 0.0038, "grad_norm": 3.4535627365112305, "learning_rate": 7.754545454545455e-06, "num_tokens": 1606047.0, "completions/mean_length": 38.0, "completions/min_length": 38.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9920530319213867, "rewards/meter/std": 0.0001019411429297179, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9920530319213867, "rewards/total_composite/std": 0.0001019411429297179, "reward": 0.9920530319213867, "reward_std": 0.00010192910121986642, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010782543569803238, "sampling/sampling_logp_difference/max": 1.0382885932922363, "sampling/importance_sampling_ratio/min": 0.3540600836277008, "sampling/importance_sampling_ratio/mean": 0.9964872002601624, "sampling/importance_sampling_ratio/max": 1.2188667058944702, "entropy": 0.055409045657143, "clip_ratio/low_mean": 0.003289473708719015, "clip_ratio/low_min": 0.003289473708719015, "clip_ratio/high_mean": 0.003289473708719015, "clip_ratio/high_max": 0.003289473708719015, "clip_ratio/region_mean": 0.00657894741743803, "reward_total_mean": 0.9920530319213867, "reward_meter_mean": 0.9920530319213867, "reward_meter_std": 0.0001019411429297179, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9920530319213867, "reward_total_composite_std": 0.0001019411429297179, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 742.0} {"timestamp_utc": "2026-04-11T20:58:34Z", "mode": "train", "global_step": 743, "epoch": 0.028691689836268148, "loss": 0.0006, "grad_norm": 0.9210623502731323, "learning_rate": 7.751515151515153e-06, "num_tokens": 1607871.0, "completions/mean_length": 62.0, "completions/min_length": 62.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9951733946800232, "rewards/meter/std": 2.3012862584437244e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9951733946800232, "rewards/total_composite/std": 2.3012862584437244e-05, "reward": 0.9951733946800232, "reward_std": 2.3008749849395826e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005985298193991184, "sampling/sampling_logp_difference/max": 0.5894505977630615, "sampling/importance_sampling_ratio/min": 0.5546319484710693, "sampling/importance_sampling_ratio/mean": 1.0004349946975708, "sampling/importance_sampling_ratio/max": 1.2164032459259033, "entropy": 0.037820551777258515, "clip_ratio/low_mean": 0.002016128972172737, "clip_ratio/low_min": 0.002016128972172737, "clip_ratio/high_mean": 0.002016128972172737, "clip_ratio/high_max": 0.002016128972172737, "clip_ratio/region_mean": 0.004032257944345474, "reward_total_mean": 0.9951733946800232, "reward_meter_mean": 0.9951733946800232, "reward_meter_std": 2.3012862584437244e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9951733946800232, "reward_total_composite_std": 2.3012862584437244e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 743.0} {"timestamp_utc": "2026-04-11T20:58:39Z", "mode": "train", "global_step": 744, "epoch": 0.028730305838739572, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.74848484848485e-06, "num_tokens": 1609735.0, "completions/mean_length": 73.0, "completions/min_length": 73.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9972173571586609, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9972173571586609, "rewards/total_composite/std": 0.0, "reward": 0.9972173571586609, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.002023279434069991, "sampling/sampling_logp_difference/max": 0.10811643302440643, "sampling/importance_sampling_ratio/min": 0.8975231051445007, "sampling/importance_sampling_ratio/mean": 1.00128972530365, "sampling/importance_sampling_ratio/max": 1.1002933979034424, "entropy": 0.02102710772305727, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9972173571586609, "reward_meter_mean": 0.9972173571586609, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9972173571586609, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 744.0} {"timestamp_utc": "2026-04-11T20:58:44Z", "mode": "train", "global_step": 745, "epoch": 0.028768921841211, "loss": 0.0252, "grad_norm": 3.4781391620635986, "learning_rate": 7.745454545454545e-06, "num_tokens": 1611457.0, "completions/mean_length": 63.25, "completions/min_length": 60.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.25, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9991881251335144, "rewards/meter/std": 0.0004611056065186858, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9991881251335144, "rewards/total_composite/std": 0.0004611056065186858, "reward": 0.9991881251335144, "reward_std": 0.000461115560028702, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010843008756637573, "sampling/sampling_logp_difference/max": 1.0690970420837402, "sampling/importance_sampling_ratio/min": 0.343318372964859, "sampling/importance_sampling_ratio/mean": 1.0012335777282715, "sampling/importance_sampling_ratio/max": 1.8684993982315063, "entropy": 0.02689387509599328, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.012231182772666216, "clip_ratio/high_max": 0.012231182772666216, "clip_ratio/region_mean": 0.01601906167343259, "reward_total_mean": 0.9991881251335144, "reward_meter_mean": 0.9991881251335144, "reward_meter_std": 0.0004611056065186858, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9991881251335144, "reward_total_composite_std": 0.0004611056065186858, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 745.0} {"timestamp_utc": "2026-04-11T20:58:52Z", "mode": "train", "global_step": 746, "epoch": 0.028807537843682424, "loss": -0.0225, "grad_norm": 0.6002373695373535, "learning_rate": 7.742424242424244e-06, "num_tokens": 1615926.0, "completions/mean_length": 333.625, "completions/min_length": 305.0, "completions/max_length": 377.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 333.625, "completions/min_terminated_length": 305.0, "completions/max_terminated_length": 377.0, "rewards/meter/mean": 0.9971553087234497, "rewards/meter/std": 0.0015759279485791922, "rewards/count_adherence/mean": 0.9444444179534912, "rewards/count_adherence/std": 0.059391383081674576, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9418157339096069, "rewards/total_composite/std": 0.060285523533821106, "reward": 0.9418157339096069, "reward_std": 0.060285523533821106, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002227205317467451, "sampling/sampling_logp_difference/max": 1.2332897186279297, "sampling/importance_sampling_ratio/min": 0.2913326025009155, "sampling/importance_sampling_ratio/mean": 1.0002700090408325, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.004979313904186711, "clip_ratio/low_mean": 0.0011247022193856537, "clip_ratio/low_min": 0.0011247022193856537, "clip_ratio/high_mean": 0.0007414010469801724, "clip_ratio/high_max": 0.0007414010469801724, "clip_ratio/region_mean": 0.0018661032663658261, "reward_total_mean": 0.9418157339096069, "reward_meter_mean": 0.9971553087234497, "reward_meter_std": 0.0015759279485791922, "reward_count_adherence_mean": 0.9444444179534912, "reward_count_adherence_std": 0.059391383081674576, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9418157339096069, "reward_total_composite_std": 0.060285523533821106, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 746.0} {"timestamp_utc": "2026-04-11T20:58:57Z", "mode": "train", "global_step": 747, "epoch": 0.028846153846153848, "loss": -0.0176, "grad_norm": 5.486899375915527, "learning_rate": 7.73939393939394e-06, "num_tokens": 1617562.0, "completions/mean_length": 47.5, "completions/min_length": 45.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.5, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.12424957752227783, "rewards/meter/std": 0.13909363746643066, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.12424957752227783, "rewards/total_composite/std": 0.13909363746643066, "reward": 0.12424957752227783, "reward_std": 0.13909363746643066, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04566996544599533, "sampling/sampling_logp_difference/max": 1.8886003494262695, "sampling/importance_sampling_ratio/min": 0.1512834131717682, "sampling/importance_sampling_ratio/mean": 0.99425208568573, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14339369256049395, "clip_ratio/low_mean": 0.03448581579141319, "clip_ratio/low_min": 0.03448581579141319, "clip_ratio/high_mean": 0.014423076994717121, "clip_ratio/high_max": 0.014423076994717121, "clip_ratio/region_mean": 0.04890889278613031, "reward_total_mean": 0.12424957752227783, "reward_meter_mean": 0.12424957752227783, "reward_meter_std": 0.13909363746643066, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.12424957752227783, "reward_total_composite_std": 0.13909363746643066, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 747.0} {"timestamp_utc": "2026-04-11T20:59:04Z", "mode": "train", "global_step": 748, "epoch": 0.028884769848625272, "loss": -0.0136, "grad_norm": 3.163942337036133, "learning_rate": 7.736363636363637e-06, "num_tokens": 1620595.0, "completions/mean_length": 180.125, "completions/min_length": 172.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 180.125, "completions/min_terminated_length": 172.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9980754852294922, "rewards/meter/std": 0.001025784877128899, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9731009006500244, "rewards/total_composite/std": 0.07027798891067505, "reward": 0.9731009006500244, "reward_std": 0.07027797400951385, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0075340294279158115, "sampling/sampling_logp_difference/max": 1.2234584093093872, "sampling/importance_sampling_ratio/min": 0.29421091079711914, "sampling/importance_sampling_ratio/mean": 1.0008153915405273, "sampling/importance_sampling_ratio/max": 1.774755597114563, "entropy": 0.03437125636264682, "clip_ratio/low_mean": 0.0007267441833391786, "clip_ratio/low_min": 0.0007267441833391786, "clip_ratio/high_mean": 0.006177731556817889, "clip_ratio/high_max": 0.006177731556817889, "clip_ratio/region_mean": 0.006904475740157068, "reward_total_mean": 0.9731009006500244, "reward_meter_mean": 0.9980754852294922, "reward_meter_std": 0.001025784877128899, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9731009006500244, "reward_total_composite_std": 0.07027798891067505, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 748.0} {"timestamp_utc": "2026-04-11T20:59:09Z", "mode": "train", "global_step": 749, "epoch": 0.028923385851096696, "loss": 0.0084, "grad_norm": 2.1292591094970703, "learning_rate": 7.733333333333334e-06, "num_tokens": 1622734.0, "completions/mean_length": 92.375, "completions/min_length": 92.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.375, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9994947910308838, "rewards/meter/std": 8.69700379553251e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9994947910308838, "rewards/total_composite/std": 8.69700379553251e-05, "reward": 0.9994947910308838, "reward_std": 8.695496944710612e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0027370797470211983, "sampling/sampling_logp_difference/max": 0.4100522994995117, "sampling/importance_sampling_ratio/min": 0.767000675201416, "sampling/importance_sampling_ratio/mean": 1.0021612644195557, "sampling/importance_sampling_ratio/max": 1.5068966150283813, "entropy": 0.013179559318814427, "clip_ratio/low_mean": 0.0013157895300537348, "clip_ratio/low_min": 0.0013157895300537348, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0013157895300537348, "reward_total_mean": 0.9994947910308838, "reward_meter_mean": 0.9994947910308838, "reward_meter_std": 8.69700379553251e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9994947910308838, "reward_total_composite_std": 8.69700379553251e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 749.0} {"timestamp_utc": "2026-04-11T20:59:19Z", "mode": "train", "global_step": 750, "epoch": 0.02896200185356812, "loss": -0.112, "grad_norm": 0.4853987693786621, "learning_rate": 7.730303030303032e-06, "num_tokens": 1624313.0, "completions/mean_length": 161.375, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.8530751466751099, "rewards/meter/std": 0.3454289436340332, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.7389551401138306, "rewards/total_composite/std": 0.45609503984451294, "reward": 0.7389551401138306, "reward_std": 0.45609503984451294, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012222505174577236, "sampling/sampling_logp_difference/max": 1.2043898105621338, "sampling/importance_sampling_ratio/min": 0.8547747731208801, "sampling/importance_sampling_ratio/mean": 1.0067731142044067, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04199734842404723, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0028409091755747795, "clip_ratio/high_max": 0.0028409091755747795, "clip_ratio/region_mean": 0.0028409091755747795, "reward_total_mean": 0.7389551401138306, "reward_meter_mean": 0.8530751466751099, "reward_meter_std": 0.3454289436340332, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.7389551401138306, "reward_total_composite_std": 0.45609503984451294, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 750.0} {"timestamp_utc": "2026-04-11T21:00:44Z", "mode": "eval", "global_step": 750, "epoch": 0.02896200185356812, "eval_loss": NaN, "eval_runtime": 85.5727, "eval_samples_per_second": 1.215, "eval_steps_per_second": 0.152, "eval_num_tokens": 1624313.0, "eval_completions/mean_length": 228.28846153846155, "eval_completions/min_length": 56.15384615384615, "eval_completions/max_length": 463.15384615384613, "eval_completions/clipped_ratio": 0.10576923076923077, "eval_completions/mean_terminated_length": 195.77701979417068, "eval_completions/min_terminated_length": 56.15384615384615, "eval_completions/max_terminated_length": 392.7692307692308, "eval_rewards/meter/mean": 0.6960660815238953, "eval_rewards/meter/std": 0.42408225857294524, "eval_rewards/count_adherence/mean": 0.915538700727316, "eval_rewards/count_adherence/std": 0.14574375748634338, "eval_rewards/arabic_clean/mean": 0.9230769230769231, "eval_rewards/arabic_clean/std": 0.16121822595596313, "eval_rewards/total_composite/mean": 0.6441489389309516, "eval_rewards/total_composite/std": 0.4322906915958111, "eval_reward": 0.6441489389309516, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0041070227833608025, "eval_sampling/sampling_logp_difference/max": 0.6373619574766892, "eval_sampling/importance_sampling_ratio/min": 0.5520345041385064, "eval_sampling/importance_sampling_ratio/mean": 1.0010583400726318, "eval_sampling/importance_sampling_ratio/max": 1.2868984937667847, "eval_entropy": 0.03480341894408831, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6441489389309516, "eval_reward_meter_mean": 0.6960660815238953, "eval_reward_meter_std": 0.42408225857294524, "eval_reward_count_adherence_mean": 0.915538700727316, "eval_reward_count_adherence_std": 0.14574375748634338, "eval_reward_arabic_clean_mean": 0.9230769230769231, "eval_reward_arabic_clean_std": 0.16121822595596313, "eval_reward_total_composite_mean": 0.6441489389309516, "eval_reward_total_composite_std": 0.4322906915958111, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 750.0} {"timestamp_utc": "2026-04-11T21:00:51Z", "mode": "train", "global_step": 751, "epoch": 0.029000617856039544, "loss": -0.0049, "grad_norm": 5.221107482910156, "learning_rate": 7.727272727272727e-06, "num_tokens": 1625755.0, "completions/mean_length": 27.25, "completions/min_length": 27.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.25, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9948364496231079, "rewards/meter/std": 0.000173440741491504, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9948364496231079, "rewards/total_composite/std": 0.000173440741491504, "reward": 0.9948364496231079, "reward_std": 0.00017344072693958879, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015429068356752396, "sampling/sampling_logp_difference/max": 0.6121819019317627, "sampling/importance_sampling_ratio/min": 0.5421666502952576, "sampling/importance_sampling_ratio/mean": 1.0034222602844238, "sampling/importance_sampling_ratio/max": 1.2863551378250122, "entropy": 0.07730915956199169, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/high_mean": 0.004464285913854837, "clip_ratio/high_max": 0.004464285913854837, "clip_ratio/region_mean": 0.009093915577977896, "reward_total_mean": 0.9948364496231079, "reward_meter_mean": 0.9948364496231079, "reward_meter_std": 0.000173440741491504, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9948364496231079, "reward_total_composite_std": 0.000173440741491504, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 751.0} {"timestamp_utc": "2026-04-11T21:00:56Z", "mode": "train", "global_step": 752, "epoch": 0.02903923385851097, "loss": -0.0227, "grad_norm": 6.488260269165039, "learning_rate": 7.724242424242424e-06, "num_tokens": 1627341.0, "completions/mean_length": 47.25, "completions/min_length": 42.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.29693859815597534, "rewards/meter/std": 0.3039378523826599, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.29693859815597534, "rewards/total_composite/std": 0.3039378523826599, "reward": 0.29693859815597534, "reward_std": 0.3039378523826599, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04617517814040184, "sampling/sampling_logp_difference/max": 1.6406193971633911, "sampling/importance_sampling_ratio/min": 0.19385991990566254, "sampling/importance_sampling_ratio/mean": 0.9970717430114746, "sampling/importance_sampling_ratio/max": 1.8714368343353271, "entropy": 0.14316083118319511, "clip_ratio/low_mean": 0.026873520808294415, "clip_ratio/low_min": 0.026873520808294415, "clip_ratio/high_mean": 0.015445990022271872, "clip_ratio/high_max": 0.015445990022271872, "clip_ratio/region_mean": 0.04231951083056629, "reward_total_mean": 0.29693859815597534, "reward_meter_mean": 0.29693859815597534, "reward_meter_std": 0.3039378523826599, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.29693859815597534, "reward_total_composite_std": 0.3039378523826599, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 752.0} {"timestamp_utc": "2026-04-11T21:01:04Z", "mode": "train", "global_step": 753, "epoch": 0.029077849860982392, "loss": 0.0001, "grad_norm": 0.21969331800937653, "learning_rate": 7.721212121212122e-06, "num_tokens": 1631846.0, "completions/mean_length": 332.125, "completions/min_length": 331.0, "completions/max_length": 334.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 332.125, "completions/min_terminated_length": 331.0, "completions/max_terminated_length": 334.0, "rewards/meter/mean": 0.9994718432426453, "rewards/meter/std": 2.581198714324273e-05, "rewards/count_adherence/mean": 0.8461538553237915, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8457069396972656, "rewards/total_composite/std": 2.185315497627016e-05, "reward": 0.8457069396972656, "reward_std": 2.1844112779945135e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0014100454282015562, "sampling/sampling_logp_difference/max": 1.4238777160644531, "sampling/importance_sampling_ratio/min": 0.24077855050563812, "sampling/importance_sampling_ratio/mean": 0.9993433952331543, "sampling/importance_sampling_ratio/max": 1.0636109113693237, "entropy": 0.004199855757178739, "clip_ratio/low_mean": 0.00037425151094794273, "clip_ratio/low_min": 0.00037425151094794273, "clip_ratio/high_mean": 0.00037650601007044315, "clip_ratio/high_max": 0.00037650601007044315, "clip_ratio/region_mean": 0.0007507575210183859, "reward_total_mean": 0.8457069396972656, "reward_meter_mean": 0.9994718432426453, "reward_meter_std": 2.581198714324273e-05, "reward_count_adherence_mean": 0.8461538553237915, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8457069396972656, "reward_total_composite_std": 2.185315497627016e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 753.0} {"timestamp_utc": "2026-04-11T21:01:09Z", "mode": "train", "global_step": 754, "epoch": 0.029116465863453816, "loss": 0.0089, "grad_norm": 5.196950912475586, "learning_rate": 7.718181818181819e-06, "num_tokens": 1633784.0, "completions/mean_length": 77.25, "completions/min_length": 76.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.25, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.08432213962078094, "rewards/meter/std": 0.00035453488817438483, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.08432213962078094, "rewards/total_composite/std": 0.00035453488817438483, "reward": 0.08432213962078094, "reward_std": 0.000354534451616928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027715306729078293, "sampling/sampling_logp_difference/max": 1.9192898273468018, "sampling/importance_sampling_ratio/min": 0.14671111106872559, "sampling/importance_sampling_ratio/mean": 0.9907717704772949, "sampling/importance_sampling_ratio/max": 1.7481837272644043, "entropy": 0.05607730080373585, "clip_ratio/low_mean": 0.0048493173671886325, "clip_ratio/low_min": 0.0048493173671886325, "clip_ratio/high_mean": 0.0032051282469183207, "clip_ratio/high_max": 0.0032051282469183207, "clip_ratio/region_mean": 0.008054445614106953, "reward_total_mean": 0.08432213962078094, "reward_meter_mean": 0.08432213962078094, "reward_meter_std": 0.00035453488817438483, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.08432213962078094, "reward_total_composite_std": 0.00035453488817438483, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 754.0} {"timestamp_utc": "2026-04-11T21:01:16Z", "mode": "train", "global_step": 755, "epoch": 0.02915508186592524, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.715151515151516e-06, "num_tokens": 1637274.0, "completions/mean_length": 227.25, "completions/min_length": 217.0, "completions/max_length": 235.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 227.25, "completions/min_terminated_length": 217.0, "completions/max_terminated_length": 235.0, "rewards/meter/mean": 0.9993188381195068, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9993188381195068, "rewards/total_composite/std": 0.0, "reward": 0.9993188381195068, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.004550791811197996, "sampling/sampling_logp_difference/max": 2.6855876445770264, "sampling/importance_sampling_ratio/min": 0.0681811198592186, "sampling/importance_sampling_ratio/mean": 0.9994167685508728, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.003425016489927657, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993188381195068, "reward_meter_mean": 0.9993188381195068, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9993188381195068, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 755.0} {"timestamp_utc": "2026-04-11T21:01:23Z", "mode": "train", "global_step": 756, "epoch": 0.029193697868396665, "loss": -0.025, "grad_norm": 0.6619568467140198, "learning_rate": 7.712121212121213e-06, "num_tokens": 1640161.0, "completions/mean_length": 174.875, "completions/min_length": 162.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 174.875, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.9930545687675476, "rewards/meter/std": 0.0004991198657080531, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9682238101959229, "rewards/total_composite/std": 0.07016286253929138, "reward": 0.9682238101959229, "reward_std": 0.07016286253929138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007540657185018063, "sampling/sampling_logp_difference/max": 1.2388986349105835, "sampling/importance_sampling_ratio/min": 0.2897031307220459, "sampling/importance_sampling_ratio/mean": 1.0004644393920898, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03254297887906432, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/high_mean": 0.00776092812884599, "clip_ratio/high_max": 0.00776092812884599, "clip_ratio/region_mean": 0.010075742960907519, "reward_total_mean": 0.9682238101959229, "reward_meter_mean": 0.9930545687675476, "reward_meter_std": 0.0004991198657080531, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9682238101959229, "reward_total_composite_std": 0.07016286253929138, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 756.0} {"timestamp_utc": "2026-04-11T21:01:27Z", "mode": "train", "global_step": 757, "epoch": 0.02923231387086809, "loss": 0.0213, "grad_norm": 2.8877339363098145, "learning_rate": 7.709090909090909e-06, "num_tokens": 1641827.0, "completions/mean_length": 54.25, "completions/min_length": 53.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.25, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9960508346557617, "rewards/meter/std": 0.0005594991962425411, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9960508346557617, "rewards/total_composite/std": 0.0005594991962425411, "reward": 0.9960508346557617, "reward_std": 0.0005594872054643929, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023436622694134712, "sampling/sampling_logp_difference/max": 1.0540428161621094, "sampling/importance_sampling_ratio/min": 0.34852585196495056, "sampling/importance_sampling_ratio/mean": 1.0017646551132202, "sampling/importance_sampling_ratio/max": 1.4908839464187622, "entropy": 0.0966068678535521, "clip_ratio/low_mean": 0.008928571827709675, "clip_ratio/low_min": 0.008928571827709675, "clip_ratio/high_mean": 0.011574073694646358, "clip_ratio/high_max": 0.011574073694646358, "clip_ratio/region_mean": 0.020502645522356033, "reward_total_mean": 0.9960508346557617, "reward_meter_mean": 0.9960508346557617, "reward_meter_std": 0.0005594991962425411, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9960508346557617, "reward_total_composite_std": 0.0005594991962425411, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 757.0} {"timestamp_utc": "2026-04-11T21:01:33Z", "mode": "train", "global_step": 758, "epoch": 0.029270929873339513, "loss": 0.0006, "grad_norm": 0.3308148980140686, "learning_rate": 7.706060606060606e-06, "num_tokens": 1644035.0, "completions/mean_length": 101.0, "completions/min_length": 101.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.0, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.0623532235622406, "rewards/meter/std": 2.084641164401546e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0623532235622406, "rewards/total_composite/std": 2.084641164401546e-05, "reward": 0.0623532235622406, "reward_std": 2.084641164401546e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004948632325977087, "sampling/sampling_logp_difference/max": 1.1194627285003662, "sampling/importance_sampling_ratio/min": 0.32645514607429504, "sampling/importance_sampling_ratio/mean": 1.0002609491348267, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.01612559356726706, "clip_ratio/low_mean": 0.002475247485563159, "clip_ratio/low_min": 0.002475247485563159, "clip_ratio/high_mean": 0.0037128712283447385, "clip_ratio/high_max": 0.0037128712283447385, "clip_ratio/region_mean": 0.0061881187139078975, "reward_total_mean": 0.0623532235622406, "reward_meter_mean": 0.0623532235622406, "reward_meter_std": 2.084641164401546e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0623532235622406, "reward_total_composite_std": 2.084641164401546e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 758.0} {"timestamp_utc": "2026-04-11T21:01:37Z", "mode": "train", "global_step": 759, "epoch": 0.029309545875810937, "loss": -0.0006, "grad_norm": 0.6779322624206543, "learning_rate": 7.703030303030304e-06, "num_tokens": 1646027.0, "completions/mean_length": 73.0, "completions/min_length": 73.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9992324113845825, "rewards/meter/std": 0.000244451715843752, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9992324113845825, "rewards/total_composite/std": 0.000244451715843752, "reward": 0.9992324113845825, "reward_std": 0.000244451715843752, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0013463424984365702, "sampling/sampling_logp_difference/max": 0.12616752088069916, "sampling/importance_sampling_ratio/min": 0.9241989254951477, "sampling/importance_sampling_ratio/mean": 1.0004380941390991, "sampling/importance_sampling_ratio/max": 1.1344722509384155, "entropy": 0.009362215059809387, "clip_ratio/low_mean": 0.0017123287543654442, "clip_ratio/low_min": 0.0017123287543654442, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0017123287543654442, "reward_total_mean": 0.9992324113845825, "reward_meter_mean": 0.9992324113845825, "reward_meter_std": 0.000244451715843752, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9992324113845825, "reward_total_composite_std": 0.000244451715843752, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 759.0} {"timestamp_utc": "2026-04-11T21:01:43Z", "mode": "train", "global_step": 760, "epoch": 0.02934816187828236, "loss": 0.0001, "grad_norm": 0.03303924947977066, "learning_rate": 7.7e-06, "num_tokens": 1648843.0, "completions/mean_length": 182.0, "completions/min_length": 182.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 182.0, "completions/min_terminated_length": 182.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.9994935393333435, "rewards/meter/std": 3.919689333997667e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9994935393333435, "rewards/total_composite/std": 3.919689333997667e-06, "reward": 0.9994935393333435, "reward_std": 3.91366347685107e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0019221891416236758, "sampling/sampling_logp_difference/max": 1.4410624504089355, "sampling/importance_sampling_ratio/min": 0.2366761565208435, "sampling/importance_sampling_ratio/mean": 0.9993479251861572, "sampling/importance_sampling_ratio/max": 1.0369774103164673, "entropy": 0.005162209854461253, "clip_ratio/low_mean": 0.0006868132040835917, "clip_ratio/low_min": 0.0006868132040835917, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0006868132040835917, "reward_total_mean": 0.9994935393333435, "reward_meter_mean": 0.9994935393333435, "reward_meter_std": 3.919689333997667e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9994935393333435, "reward_total_composite_std": 3.919689333997667e-06, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 760.0} {"timestamp_utc": "2026-04-11T21:01:53Z", "mode": "train", "global_step": 761, "epoch": 0.029386777880753785, "loss": -0.0355, "grad_norm": 10.61474895477295, "learning_rate": 7.696969696969696e-06, "num_tokens": 1650490.0, "completions/mean_length": 107.875, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 50.142860412597656, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8295968770980835, "rewards/meter/std": 0.31729573011398315, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7507349252700806, "rewards/total_composite/std": 0.4315637946128845, "reward": 0.7507349252700806, "reward_std": 0.4315637946128845, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06746242195367813, "sampling/sampling_logp_difference/max": 5.475861549377441, "sampling/importance_sampling_ratio/min": 0.004186620004475117, "sampling/importance_sampling_ratio/mean": 1.001595377922058, "sampling/importance_sampling_ratio/max": 1.5547585487365723, "entropy": 0.15587164601311088, "clip_ratio/low_mean": 0.0033333334140479565, "clip_ratio/low_min": 0.0033333334140479565, "clip_ratio/high_mean": 0.011000967118889093, "clip_ratio/high_max": 0.011000967118889093, "clip_ratio/region_mean": 0.01433430053293705, "reward_total_mean": 0.7507349252700806, "reward_meter_mean": 0.8295968770980835, "reward_meter_std": 0.31729573011398315, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7507349252700806, "reward_total_composite_std": 0.4315637946128845, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 761.0} {"timestamp_utc": "2026-04-11T21:01:58Z", "mode": "train", "global_step": 762, "epoch": 0.02942539388322521, "loss": 0.0034, "grad_norm": 5.589205741882324, "learning_rate": 7.693939393939395e-06, "num_tokens": 1652177.0, "completions/mean_length": 53.875, "completions/min_length": 53.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.15666525065898895, "rewards/meter/std": 0.0002791965380311012, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.15666525065898895, "rewards/total_composite/std": 0.0002791965380311012, "reward": 0.15666525065898895, "reward_std": 0.00027919973945245147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028472045436501503, "sampling/sampling_logp_difference/max": 2.145233154296875, "sampling/importance_sampling_ratio/min": 0.1170407384634018, "sampling/importance_sampling_ratio/mean": 0.9879463315010071, "sampling/importance_sampling_ratio/max": 1.3293702602386475, "entropy": 0.05692701740190387, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/high_mean": 0.015951178269460797, "clip_ratio/high_max": 0.015951178269460797, "clip_ratio/region_mean": 0.01830966887064278, "reward_total_mean": 0.15666525065898895, "reward_meter_mean": 0.15666525065898895, "reward_meter_std": 0.0002791965380311012, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.15666525065898895, "reward_total_composite_std": 0.0002791965380311012, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 762.0} {"timestamp_utc": "2026-04-11T21:02:03Z", "mode": "train", "global_step": 763, "epoch": 0.029464009885696633, "loss": 0.0131, "grad_norm": 6.319483280181885, "learning_rate": 7.690909090909091e-06, "num_tokens": 1653829.0, "completions/mean_length": 54.5, "completions/min_length": 53.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9953004121780396, "rewards/meter/std": 0.0016852071275934577, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9953004121780396, "rewards/total_composite/std": 0.0016852071275934577, "reward": 0.9953004121780396, "reward_std": 0.0016852213302627206, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02161485329270363, "sampling/sampling_logp_difference/max": 1.3513164520263672, "sampling/importance_sampling_ratio/min": 0.25889918208122253, "sampling/importance_sampling_ratio/mean": 1.0041148662567139, "sampling/importance_sampling_ratio/max": 1.7141997814178467, "entropy": 0.12393038254231215, "clip_ratio/low_mean": 0.01116071455180645, "clip_ratio/low_min": 0.01116071455180645, "clip_ratio/high_mean": 0.016039948211982846, "clip_ratio/high_max": 0.016039948211982846, "clip_ratio/region_mean": 0.027200662763789296, "reward_total_mean": 0.9953004121780396, "reward_meter_mean": 0.9953004121780396, "reward_meter_std": 0.0016852071275934577, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9953004121780396, "reward_total_composite_std": 0.0016852071275934577, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 763.0} {"timestamp_utc": "2026-04-11T21:02:11Z", "mode": "train", "global_step": 764, "epoch": 0.029502625888168058, "loss": 0.0042, "grad_norm": 33.3278923034668, "learning_rate": 7.687878787878788e-06, "num_tokens": 1658430.0, "completions/mean_length": 372.125, "completions/min_length": 349.0, "completions/max_length": 397.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 372.125, "completions/min_terminated_length": 349.0, "completions/max_terminated_length": 397.0, "rewards/meter/mean": 0.739122211933136, "rewards/meter/std": 0.45561060309410095, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.739122211933136, "rewards/total_composite/std": 0.45561060309410095, "reward": 0.739122211933136, "reward_std": 0.45561057329177856, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0034644489642232656, "sampling/sampling_logp_difference/max": 2.3520443439483643, "sampling/importance_sampling_ratio/min": 0.0951743945479393, "sampling/importance_sampling_ratio/mean": 0.9990094900131226, "sampling/importance_sampling_ratio/max": 1.448507308959961, "entropy": 0.007602737401612103, "clip_ratio/low_mean": 0.0003324467979837209, "clip_ratio/low_min": 0.0003324467979837209, "clip_ratio/high_mean": 0.0013256430975161493, "clip_ratio/high_max": 0.0013256430975161493, "clip_ratio/region_mean": 0.0016580898954998702, "reward_total_mean": 0.739122211933136, "reward_meter_mean": 0.739122211933136, "reward_meter_std": 0.45561060309410095, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.739122211933136, "reward_total_composite_std": 0.45561060309410095, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 764.0} {"timestamp_utc": "2026-04-11T21:02:16Z", "mode": "train", "global_step": 765, "epoch": 0.02954124189063948, "loss": -0.0174, "grad_norm": 0.7466166615486145, "learning_rate": 7.684848484848485e-06, "num_tokens": 1660261.0, "completions/mean_length": 71.875, "completions/min_length": 67.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9964892268180847, "rewards/meter/std": 0.0027039791457355022, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9964892268180847, "rewards/total_composite/std": 0.0027039791457355022, "reward": 0.9964892268180847, "reward_std": 0.002703979378566146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0036985778715461493, "sampling/sampling_logp_difference/max": 0.9250373840332031, "sampling/importance_sampling_ratio/min": 0.7690510153770447, "sampling/importance_sampling_ratio/mean": 1.0021460056304932, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.01590311119798571, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0018656715983524919, "reward_total_mean": 0.9964892268180847, "reward_meter_mean": 0.9964892268180847, "reward_meter_std": 0.0027039791457355022, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9964892268180847, "reward_total_composite_std": 0.0027039791457355022, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 765.0} {"timestamp_utc": "2026-04-11T21:02:21Z", "mode": "train", "global_step": 766, "epoch": 0.029579857893110906, "loss": -0.006, "grad_norm": 8.477374076843262, "learning_rate": 7.681818181818183e-06, "num_tokens": 1662053.0, "completions/mean_length": 61.0, "completions/min_length": 58.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.990821897983551, "rewards/meter/std": 0.023206030949950218, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.990821897983551, "rewards/total_composite/std": 0.023206030949950218, "reward": 0.990821897983551, "reward_std": 0.023206030949950218, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034518104046583176, "sampling/sampling_logp_difference/max": 3.6948211193084717, "sampling/importance_sampling_ratio/min": 0.024851899594068527, "sampling/importance_sampling_ratio/mean": 0.9888519644737244, "sampling/importance_sampling_ratio/max": 1.6260812282562256, "entropy": 0.08710672426968813, "clip_ratio/low_mean": 0.0062500000931322575, "clip_ratio/low_min": 0.0062500000931322575, "clip_ratio/high_mean": 0.026962829288095236, "clip_ratio/high_max": 0.026962829288095236, "clip_ratio/region_mean": 0.03321282938122749, "reward_total_mean": 0.990821897983551, "reward_meter_mean": 0.990821897983551, "reward_meter_std": 0.023206030949950218, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.990821897983551, "reward_total_composite_std": 0.023206030949950218, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 766.0} {"timestamp_utc": "2026-04-11T21:02:26Z", "mode": "train", "global_step": 767, "epoch": 0.02961847389558233, "loss": 0.1145, "grad_norm": 10.482832908630371, "learning_rate": 7.678787878787878e-06, "num_tokens": 1663669.0, "completions/mean_length": 48.0, "completions/min_length": 41.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.935056746006012, "rewards/meter/std": 0.11260737478733063, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.935056746006012, "rewards/total_composite/std": 0.11260737478733063, "reward": 0.935056746006012, "reward_std": 0.11260735988616943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04064905270934105, "sampling/sampling_logp_difference/max": 0.9094129800796509, "sampling/importance_sampling_ratio/min": 0.5011124610900879, "sampling/importance_sampling_ratio/mean": 1.0120450258255005, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19717669300734997, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/high_mean": 0.011037152959033847, "clip_ratio/high_max": 0.011037152959033847, "clip_ratio/region_mean": 0.015135513385757804, "reward_total_mean": 0.935056746006012, "reward_meter_mean": 0.935056746006012, "reward_meter_std": 0.11260737478733063, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.935056746006012, "reward_total_composite_std": 0.11260737478733063, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 767.0} {"timestamp_utc": "2026-04-11T21:02:32Z", "mode": "train", "global_step": 768, "epoch": 0.029657089898053754, "loss": 0.0079, "grad_norm": 3.7159321308135986, "learning_rate": 7.675757575757577e-06, "num_tokens": 1665640.0, "completions/mean_length": 70.375, "completions/min_length": 65.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.992908775806427, "rewards/meter/std": 0.0015179176116362214, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.992908775806427, "rewards/total_composite/std": 0.0015179176116362214, "reward": 0.992908775806427, "reward_std": 0.001517921919003129, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038023777306079865, "sampling/sampling_logp_difference/max": 5.909420013427734, "sampling/importance_sampling_ratio/min": 0.0027137603610754013, "sampling/importance_sampling_ratio/mean": 1.0022401809692383, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14855861850082874, "clip_ratio/low_mean": 0.008883806294761598, "clip_ratio/low_min": 0.008883806294761598, "clip_ratio/high_mean": 0.01945881894789636, "clip_ratio/high_max": 0.01945881894789636, "clip_ratio/region_mean": 0.02834262524265796, "reward_total_mean": 0.992908775806427, "reward_meter_mean": 0.992908775806427, "reward_meter_std": 0.0015179176116362214, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.992908775806427, "reward_total_composite_std": 0.0015179176116362214, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 768.0} {"timestamp_utc": "2026-04-11T21:02:37Z", "mode": "train", "global_step": 769, "epoch": 0.029695705900525178, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.672727272727273e-06, "num_tokens": 1667456.0, "completions/mean_length": 70.0, "completions/min_length": 70.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9986898303031921, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9986898303031921, "rewards/total_composite/std": 0.0, "reward": 0.9986898303031921, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0021432472858577967, "sampling/sampling_logp_difference/max": 0.29098376631736755, "sampling/importance_sampling_ratio/min": 0.7475277781486511, "sampling/importance_sampling_ratio/mean": 1.0004918575286865, "sampling/importance_sampling_ratio/max": 1.0774732828140259, "entropy": 0.020252354443073273, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9986898303031921, "reward_meter_mean": 0.9986898303031921, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9986898303031921, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 769.0} {"timestamp_utc": "2026-04-11T21:02:43Z", "mode": "train", "global_step": 770, "epoch": 0.029734321902996602, "loss": -0.0568, "grad_norm": 3.0930964946746826, "learning_rate": 7.66969696969697e-06, "num_tokens": 1669853.0, "completions/mean_length": 139.625, "completions/min_length": 120.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.625, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.9882319569587708, "rewards/meter/std": 0.00797359086573124, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9882319569587708, "rewards/total_composite/std": 0.00797359086573124, "reward": 0.9882319569587708, "reward_std": 0.007973585277795792, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02833718992769718, "sampling/sampling_logp_difference/max": 7.906123161315918, "sampling/importance_sampling_ratio/min": 0.0003684803668875247, "sampling/importance_sampling_ratio/mean": 0.9987453818321228, "sampling/importance_sampling_ratio/max": 1.6485743522644043, "entropy": 0.04922680510208011, "clip_ratio/low_mean": 0.0010416667209938169, "clip_ratio/low_min": 0.0010416667209938169, "clip_ratio/high_mean": 0.0051819164655171335, "clip_ratio/high_max": 0.0051819164655171335, "clip_ratio/region_mean": 0.00622358318651095, "reward_total_mean": 0.9882319569587708, "reward_meter_mean": 0.9882319569587708, "reward_meter_std": 0.00797359086573124, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9882319569587708, "reward_total_composite_std": 0.00797359086573124, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 770.0} {"timestamp_utc": "2026-04-11T21:02:48Z", "mode": "train", "global_step": 771, "epoch": 0.029772937905468026, "loss": 0.0025, "grad_norm": 0.2530145049095154, "learning_rate": 7.666666666666667e-06, "num_tokens": 1671469.0, "completions/mean_length": 47.0, "completions/min_length": 47.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.0, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9821916222572327, "rewards/meter/std": 0.0005827855784446001, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9821916222572327, "rewards/total_composite/std": 0.0005827855784446001, "reward": 0.9821916222572327, "reward_std": 0.0005827946006320417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004841628018766642, "sampling/sampling_logp_difference/max": 0.7497587203979492, "sampling/importance_sampling_ratio/min": 0.47248056530952454, "sampling/importance_sampling_ratio/mean": 0.9999971389770508, "sampling/importance_sampling_ratio/max": 1.0474599599838257, "entropy": 0.02513228729367256, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.002659574383869767, "clip_ratio/high_max": 0.002659574383869767, "clip_ratio/region_mean": 0.002659574383869767, "reward_total_mean": 0.9821916222572327, "reward_meter_mean": 0.9821916222572327, "reward_meter_std": 0.0005827855784446001, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9821916222572327, "reward_total_composite_std": 0.0005827855784446001, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 771.0} {"timestamp_utc": "2026-04-11T21:02:54Z", "mode": "train", "global_step": 772, "epoch": 0.02981155390793945, "loss": -0.0109, "grad_norm": 1.56611168384552, "learning_rate": 7.663636363636364e-06, "num_tokens": 1674625.0, "completions/mean_length": 187.5, "completions/min_length": 177.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 187.5, "completions/min_terminated_length": 177.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.9984389543533325, "rewards/meter/std": 0.0001413605350535363, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984389543533325, "rewards/total_composite/std": 0.0001413605350535363, "reward": 0.9984389543533325, "reward_std": 0.0001413605350535363, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002058023354038596, "sampling/sampling_logp_difference/max": 0.7111762762069702, "sampling/importance_sampling_ratio/min": 0.49106621742248535, "sampling/importance_sampling_ratio/mean": 1.0000642538070679, "sampling/importance_sampling_ratio/max": 1.5726603269577026, "entropy": 0.010261786286719143, "clip_ratio/low_mean": 0.0007062146905809641, "clip_ratio/low_min": 0.0007062146905809641, "clip_ratio/high_mean": 0.0006613756413571537, "clip_ratio/high_max": 0.0006613756413571537, "clip_ratio/region_mean": 0.0013675903319381177, "reward_total_mean": 0.9984389543533325, "reward_meter_mean": 0.9984389543533325, "reward_meter_std": 0.0001413605350535363, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984389543533325, "reward_total_composite_std": 0.0001413605350535363, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 772.0} {"timestamp_utc": "2026-04-11T21:02:58Z", "mode": "train", "global_step": 773, "epoch": 0.029850169910410874, "loss": -0.0174, "grad_norm": 8.638322830200195, "learning_rate": 7.660606060606062e-06, "num_tokens": 1676022.0, "completions/mean_length": 35.625, "completions/min_length": 33.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9978481531143188, "rewards/meter/std": 0.0021117855794727802, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9978481531143188, "rewards/total_composite/std": 0.0021117855794727802, "reward": 0.9978481531143188, "reward_std": 0.0021117855794727802, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02384253405034542, "sampling/sampling_logp_difference/max": 1.2632678747177124, "sampling/importance_sampling_ratio/min": 0.2827285826206207, "sampling/importance_sampling_ratio/mean": 1.002649188041687, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11014944780617952, "clip_ratio/low_mean": 0.014204545877873898, "clip_ratio/low_min": 0.014204545877873898, "clip_ratio/high_mean": 0.0034722222480922937, "clip_ratio/high_max": 0.0034722222480922937, "clip_ratio/region_mean": 0.01767676812596619, "reward_total_mean": 0.9978481531143188, "reward_meter_mean": 0.9978481531143188, "reward_meter_std": 0.0021117855794727802, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9978481531143188, "reward_total_composite_std": 0.0021117855794727802, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 773.0} {"timestamp_utc": "2026-04-11T21:03:03Z", "mode": "train", "global_step": 774, "epoch": 0.0298887859128823, "loss": 0.0229, "grad_norm": 19.09847068786621, "learning_rate": 7.657575757575757e-06, "num_tokens": 1677551.0, "completions/mean_length": 31.125, "completions/min_length": 31.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.996236264705658, "rewards/meter/std": 0.003699568100273609, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.996236264705658, "rewards/total_composite/std": 0.003699568100273609, "reward": 0.996236264705658, "reward_std": 0.003699559485539794, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029653802514076233, "sampling/sampling_logp_difference/max": 0.9775059819221497, "sampling/importance_sampling_ratio/min": 0.376248300075531, "sampling/importance_sampling_ratio/mean": 1.0019131898880005, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0747169773094356, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/high_mean": 0.016129031777381897, "clip_ratio/high_max": 0.016129031777381897, "clip_ratio/region_mean": 0.02016128972172737, "reward_total_mean": 0.996236264705658, "reward_meter_mean": 0.996236264705658, "reward_meter_std": 0.003699568100273609, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.996236264705658, "reward_total_composite_std": 0.003699568100273609, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 774.0} {"timestamp_utc": "2026-04-11T21:03:13Z", "mode": "train", "global_step": 775, "epoch": 0.029927401915353723, "loss": 0.0345, "grad_norm": 6.605960369110107, "learning_rate": 7.654545454545456e-06, "num_tokens": 1678954.0, "completions/mean_length": 91.375, "completions/min_length": 20.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 31.285715103149414, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.1142534390091896, "rewards/meter/std": 0.3230631947517395, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.1142534390091896, "rewards/total_composite/std": 0.3230631947517395, "reward": 0.1142534390091896, "reward_std": 0.3230631947517395, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08634916692972183, "sampling/sampling_logp_difference/max": 1.715205192565918, "sampling/importance_sampling_ratio/min": 0.1799268126487732, "sampling/importance_sampling_ratio/mean": 0.9676999449729919, "sampling/importance_sampling_ratio/max": 1.5792129039764404, "entropy": 0.18733000196516514, "clip_ratio/low_mean": 0.018723540706560016, "clip_ratio/low_min": 0.018723540706560016, "clip_ratio/high_mean": 0.03750000149011612, "clip_ratio/high_max": 0.03750000149011612, "clip_ratio/region_mean": 0.056223542196676135, "reward_total_mean": 0.1142534390091896, "reward_meter_mean": 0.1142534390091896, "reward_meter_std": 0.3230631947517395, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.1142534390091896, "reward_total_composite_std": 0.3230631947517395, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 775.0} {"timestamp_utc": "2026-04-11T21:03:18Z", "mode": "train", "global_step": 776, "epoch": 0.029966017917825147, "loss": -0.0018, "grad_norm": 9.811491012573242, "learning_rate": 7.651515151515152e-06, "num_tokens": 1680713.0, "completions/mean_length": 64.875, "completions/min_length": 47.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.8518927097320557, "rewards/meter/std": 0.313870370388031, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7919032573699951, "rewards/total_composite/std": 0.3354162871837616, "reward": 0.7919032573699951, "reward_std": 0.3354162871837616, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0203444454818964, "sampling/sampling_logp_difference/max": 1.3591630458831787, "sampling/importance_sampling_ratio/min": 0.2568756639957428, "sampling/importance_sampling_ratio/mean": 0.9963316917419434, "sampling/importance_sampling_ratio/max": 1.88141667842865, "entropy": 0.06882636761292815, "clip_ratio/low_mean": 0.0013297871919348836, "clip_ratio/low_min": 0.0013297871919348836, "clip_ratio/high_mean": 0.013776532025076449, "clip_ratio/high_max": 0.013776532025076449, "clip_ratio/region_mean": 0.015106319217011333, "reward_total_mean": 0.7919032573699951, "reward_meter_mean": 0.8518927097320557, "reward_meter_std": 0.313870370388031, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7919032573699951, "reward_total_composite_std": 0.3354162871837616, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 776.0} {"timestamp_utc": "2026-04-11T21:03:23Z", "mode": "train", "global_step": 777, "epoch": 0.03000463392029657, "loss": 0.0179, "grad_norm": 4.039414882659912, "learning_rate": 7.648484848484849e-06, "num_tokens": 1682615.0, "completions/mean_length": 70.75, "completions/min_length": 69.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.75, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9934273362159729, "rewards/meter/std": 0.001961036818102002, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9934273362159729, "rewards/total_composite/std": 0.001961036818102002, "reward": 0.9934273362159729, "reward_std": 0.0019610452000051737, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00839174259454012, "sampling/sampling_logp_difference/max": 0.45795851945877075, "sampling/importance_sampling_ratio/min": 0.6325737237930298, "sampling/importance_sampling_ratio/mean": 1.0014435052871704, "sampling/importance_sampling_ratio/max": 1.3909709453582764, "entropy": 0.04455849016085267, "clip_ratio/low_mean": 0.0035239229910075665, "clip_ratio/low_min": 0.0035239229910075665, "clip_ratio/high_mean": 0.005357142887078226, "clip_ratio/high_max": 0.005357142887078226, "clip_ratio/region_mean": 0.008881065878085792, "reward_total_mean": 0.9934273362159729, "reward_meter_mean": 0.9934273362159729, "reward_meter_std": 0.001961036818102002, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9934273362159729, "reward_total_composite_std": 0.001961036818102002, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 777.0} {"timestamp_utc": "2026-04-11T21:03:28Z", "mode": "train", "global_step": 778, "epoch": 0.030043249922767995, "loss": 0.0031, "grad_norm": 1.1315220594406128, "learning_rate": 7.645454545454546e-06, "num_tokens": 1684509.0, "completions/mean_length": 70.75, "completions/min_length": 70.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.75, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9986703395843506, "rewards/meter/std": 4.082572559127584e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9986703395843506, "rewards/total_composite/std": 4.082572559127584e-05, "reward": 0.9986703395843506, "reward_std": 4.082230952917598e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007064300123602152, "sampling/sampling_logp_difference/max": 1.7487998008728027, "sampling/importance_sampling_ratio/min": 0.1739826202392578, "sampling/importance_sampling_ratio/mean": 0.9994363784790039, "sampling/importance_sampling_ratio/max": 1.2364037036895752, "entropy": 0.022018407587893307, "clip_ratio/low_mean": 0.0034246575087308884, "clip_ratio/low_min": 0.0034246575087308884, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0034246575087308884, "reward_total_mean": 0.9986703395843506, "reward_meter_mean": 0.9986703395843506, "reward_meter_std": 4.082572559127584e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9986703395843506, "reward_total_composite_std": 4.082572559127584e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 778.0} {"timestamp_utc": "2026-04-11T21:03:34Z", "mode": "train", "global_step": 779, "epoch": 0.03008186592523942, "loss": -0.0001, "grad_norm": 0.3496533930301666, "learning_rate": 7.642424242424244e-06, "num_tokens": 1687669.0, "completions/mean_length": 181.0, "completions/min_length": 181.0, "completions/max_length": 181.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.0, "completions/min_terminated_length": 181.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.9982551336288452, "rewards/meter/std": 3.099746027146466e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9982551336288452, "rewards/total_composite/std": 3.099746027146466e-05, "reward": 0.9982551336288452, "reward_std": 3.10054820147343e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0030267720576375723, "sampling/sampling_logp_difference/max": 0.40749502182006836, "sampling/importance_sampling_ratio/min": 0.6653147339820862, "sampling/importance_sampling_ratio/mean": 1.000686526298523, "sampling/importance_sampling_ratio/max": 1.4227917194366455, "entropy": 0.015219439752399921, "clip_ratio/low_mean": 0.002071823284495622, "clip_ratio/low_min": 0.002071823284495622, "clip_ratio/high_mean": 0.0006906077614985406, "clip_ratio/high_max": 0.0006906077614985406, "clip_ratio/region_mean": 0.0027624310459941626, "reward_total_mean": 0.9982551336288452, "reward_meter_mean": 0.9982551336288452, "reward_meter_std": 3.099746027146466e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9982551336288452, "reward_total_composite_std": 3.099746027146466e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 779.0} {"timestamp_utc": "2026-04-11T21:03:41Z", "mode": "train", "global_step": 780, "epoch": 0.030120481927710843, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.639393939393939e-06, "num_tokens": 1691133.0, "completions/mean_length": 238.0, "completions/min_length": 238.0, "completions/max_length": 238.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 238.0, "completions/min_terminated_length": 238.0, "completions/max_terminated_length": 238.0, "rewards/meter/mean": 0.9987215995788574, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9987215995788574, "rewards/total_composite/std": 0.0, "reward": 0.9987215995788574, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006873520324006677, "sampling/sampling_logp_difference/max": 0.18778496980667114, "sampling/importance_sampling_ratio/min": 0.828792929649353, "sampling/importance_sampling_ratio/mean": 1.0000550746917725, "sampling/importance_sampling_ratio/max": 1.037010908126831, "entropy": 0.005387227836763486, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9987215995788574, "reward_meter_mean": 0.9987215995788574, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9987215995788574, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 780.0} {"timestamp_utc": "2026-04-11T21:03:51Z", "mode": "train", "global_step": 781, "epoch": 0.030159097930182267, "loss": -0.0097, "grad_norm": 3.0829520225524902, "learning_rate": 7.636363636363638e-06, "num_tokens": 1696693.0, "completions/mean_length": 463.0, "completions/min_length": 449.0, "completions/max_length": 465.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 463.0, "completions/min_terminated_length": 449.0, "completions/max_terminated_length": 465.0, "rewards/meter/mean": 0.8724232912063599, "rewards/meter/std": 0.3490966856479645, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.0235702246427536, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8141912221908569, "rewards/total_composite/std": 0.32602277398109436, "reward": 0.8141912221908569, "reward_std": 0.32602280378341675, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0017965902807191014, "sampling/sampling_logp_difference/max": 1.6903929710388184, "sampling/importance_sampling_ratio/min": 0.6545984745025635, "sampling/importance_sampling_ratio/mean": 1.0003784894943237, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.004711298577603884, "clip_ratio/low_mean": 0.000835189304780215, "clip_ratio/low_min": 0.000835189304780215, "clip_ratio/high_mean": 0.0005376344197429717, "clip_ratio/high_max": 0.0005376344197429717, "clip_ratio/region_mean": 0.0013728237245231867, "reward_total_mean": 0.8141912221908569, "reward_meter_mean": 0.8724232912063599, "reward_meter_std": 0.3490966856479645, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.0235702246427536, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8141912221908569, "reward_total_composite_std": 0.32602277398109436, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 781.0} {"timestamp_utc": "2026-04-11T21:03:56Z", "mode": "train", "global_step": 782, "epoch": 0.03019771393265369, "loss": 0.0041, "grad_norm": 4.670520305633545, "learning_rate": 7.633333333333334e-06, "num_tokens": 1698520.0, "completions/mean_length": 68.375, "completions/min_length": 66.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9923343062400818, "rewards/meter/std": 0.007388563361018896, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9923343062400818, "rewards/total_composite/std": 0.007388563361018896, "reward": 0.9923343062400818, "reward_std": 0.007388569414615631, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013357391580939293, "sampling/sampling_logp_difference/max": 0.9248440265655518, "sampling/importance_sampling_ratio/min": 0.396593302488327, "sampling/importance_sampling_ratio/mean": 0.9997692704200745, "sampling/importance_sampling_ratio/max": 1.6508756875991821, "entropy": 0.06113292137160897, "clip_ratio/low_mean": 0.005434782709926367, "clip_ratio/low_min": 0.005434782709926367, "clip_ratio/high_mean": 0.009111253311857581, "clip_ratio/high_max": 0.009111253311857581, "clip_ratio/region_mean": 0.014546036021783948, "reward_total_mean": 0.9923343062400818, "reward_meter_mean": 0.9923343062400818, "reward_meter_std": 0.007388563361018896, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9923343062400818, "reward_total_composite_std": 0.007388563361018896, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 782.0} {"timestamp_utc": "2026-04-11T21:04:00Z", "mode": "train", "global_step": 783, "epoch": 0.030236329935125116, "loss": 0.0085, "grad_norm": 3.8934073448181152, "learning_rate": 7.630303030303031e-06, "num_tokens": 1700044.0, "completions/mean_length": 36.5, "completions/min_length": 35.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9990229606628418, "rewards/meter/std": 0.00019247460295446217, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9990229606628418, "rewards/total_composite/std": 0.00019247460295446217, "reward": 0.9990229606628418, "reward_std": 0.00019248879107180983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023439016193151474, "sampling/sampling_logp_difference/max": 3.9412550926208496, "sampling/importance_sampling_ratio/min": 0.019423820078372955, "sampling/importance_sampling_ratio/mean": 0.9912862777709961, "sampling/importance_sampling_ratio/max": 1.057726502418518, "entropy": 0.03145878529176116, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.010248779086396098, "clip_ratio/high_max": 0.010248779086396098, "clip_ratio/region_mean": 0.010248779086396098, "reward_total_mean": 0.9990229606628418, "reward_meter_mean": 0.9990229606628418, "reward_meter_std": 0.00019247460295446217, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9990229606628418, "reward_total_composite_std": 0.00019247460295446217, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 783.0} {"timestamp_utc": "2026-04-11T21:04:10Z", "mode": "train", "global_step": 784, "epoch": 0.03027494593759654, "loss": -0.2065, "grad_norm": 0.864739179611206, "learning_rate": 7.627272727272727e-06, "num_tokens": 1702195.0, "completions/mean_length": 155.875, "completions/min_length": 104.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 105.00000762939453, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9695366621017456, "rewards/meter/std": 0.03671129420399666, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8592759370803833, "rewards/total_composite/std": 0.3473426401615143, "reward": 0.8592759370803833, "reward_std": 0.3473426103591919, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017855394631624222, "sampling/sampling_logp_difference/max": 2.2580552101135254, "sampling/importance_sampling_ratio/min": 0.10455362498760223, "sampling/importance_sampling_ratio/mean": 0.9962214827537537, "sampling/importance_sampling_ratio/max": 1.6998333930969238, "entropy": 0.03899357304908335, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.014244775520637631, "clip_ratio/high_max": 0.014244775520637631, "clip_ratio/region_mean": 0.014244775520637631, "reward_total_mean": 0.8592759370803833, "reward_meter_mean": 0.9695366621017456, "reward_meter_std": 0.03671129420399666, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8592759370803833, "reward_total_composite_std": 0.3473426401615143, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 784.0} {"timestamp_utc": "2026-04-11T21:04:16Z", "mode": "train", "global_step": 785, "epoch": 0.030313561940067964, "loss": 0.0208, "grad_norm": 5.147727966308594, "learning_rate": 7.6242424242424254e-06, "num_tokens": 1704684.0, "completions/mean_length": 122.125, "completions/min_length": 119.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.125, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.8801974058151245, "rewards/meter/std": 0.24488385021686554, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8801974058151245, "rewards/total_composite/std": 0.24488385021686554, "reward": 0.8801974058151245, "reward_std": 0.24488388001918793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.037598490715026855, "sampling/sampling_logp_difference/max": 13.08521556854248, "sampling/importance_sampling_ratio/min": 2.0756926915055374e-06, "sampling/importance_sampling_ratio/mean": 1.0013856887817383, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11644519306719303, "clip_ratio/low_mean": 0.010881452355533838, "clip_ratio/low_min": 0.010881452355533838, "clip_ratio/high_mean": 0.010341346263885498, "clip_ratio/high_max": 0.010341346263885498, "clip_ratio/region_mean": 0.021222798619419336, "reward_total_mean": 0.8801974058151245, "reward_meter_mean": 0.8801974058151245, "reward_meter_std": 0.24488385021686554, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8801974058151245, "reward_total_composite_std": 0.24488385021686554, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 785.0} {"timestamp_utc": "2026-04-11T21:04:25Z", "mode": "train", "global_step": 786, "epoch": 0.030352177942539388, "loss": -0.1731, "grad_norm": 1.139194369316101, "learning_rate": 7.621212121212122e-06, "num_tokens": 1706457.0, "completions/mean_length": 126.625, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 71.5714340209961, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.8664183616638184, "rewards/meter/std": 0.33893653750419617, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8629486560821533, "rewards/total_composite/std": 0.3487482964992523, "reward": 0.8629486560821533, "reward_std": 0.34874826669692993, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030048077926039696, "sampling/sampling_logp_difference/max": 3.611645460128784, "sampling/importance_sampling_ratio/min": 0.027007373049855232, "sampling/importance_sampling_ratio/mean": 0.9959703087806702, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03290584729984403, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007095410721376538, "clip_ratio/high_max": 0.007095410721376538, "clip_ratio/region_mean": 0.007095410721376538, "reward_total_mean": 0.8629486560821533, "reward_meter_mean": 0.8664183616638184, "reward_meter_std": 0.33893653750419617, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8629486560821533, "reward_total_composite_std": 0.3487482964992523, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 786.0} {"timestamp_utc": "2026-04-11T21:04:35Z", "mode": "train", "global_step": 787, "epoch": 0.030390793945010812, "loss": -0.1075, "grad_norm": 1.1968501806259155, "learning_rate": 7.618181818181819e-06, "num_tokens": 1707861.0, "completions/mean_length": 225.5, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 53.60000228881836, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8516892194747925, "rewards/meter/std": 0.3451445698738098, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.37796446681022644, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.6139416694641113, "rewards/total_composite/std": 0.5088551044464111, "reward": 0.6139416694641113, "reward_std": 0.5088551044464111, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03149334713816643, "sampling/sampling_logp_difference/max": 1.614349365234375, "sampling/importance_sampling_ratio/min": 0.3030705451965332, "sampling/importance_sampling_ratio/mean": 1.0082732439041138, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09022841136902571, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.01620688010007143, "clip_ratio/high_max": 0.01620688010007143, "clip_ratio/region_mean": 0.01620688010007143, "reward_total_mean": 0.6139416694641113, "reward_meter_mean": 0.8516892194747925, "reward_meter_std": 0.3451445698738098, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.37796446681022644, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.6139416694641113, "reward_total_composite_std": 0.5088551044464111, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 787.0} {"timestamp_utc": "2026-04-11T21:04:45Z", "mode": "train", "global_step": 788, "epoch": 0.030429409947482236, "loss": -0.064, "grad_norm": 1.5179235935211182, "learning_rate": 7.6151515151515155e-06, "num_tokens": 1709402.0, "completions/mean_length": 106.625, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 48.71428680419922, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7512721419334412, "rewards/meter/std": 0.34811529517173767, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6802998781204224, "rewards/total_composite/std": 0.4373186528682709, "reward": 0.6802998781204224, "reward_std": 0.4373186528682709, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039495352655649185, "sampling/sampling_logp_difference/max": 1.08758544921875, "sampling/importance_sampling_ratio/min": 0.33702927827835083, "sampling/importance_sampling_ratio/mean": 1.0152835845947266, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18782542645931244, "clip_ratio/low_mean": 0.009259259328246117, "clip_ratio/low_min": 0.009259259328246117, "clip_ratio/high_mean": 0.010529743740335107, "clip_ratio/high_max": 0.010529743740335107, "clip_ratio/region_mean": 0.019789003068581223, "reward_total_mean": 0.6802998781204224, "reward_meter_mean": 0.7512721419334412, "reward_meter_std": 0.34811529517173767, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6802998781204224, "reward_total_composite_std": 0.4373186528682709, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 788.0} {"timestamp_utc": "2026-04-11T21:04:54Z", "mode": "train", "global_step": 789, "epoch": 0.03046802594995366, "loss": -0.1734, "grad_norm": 0.28587642312049866, "learning_rate": 7.612121212121213e-06, "num_tokens": 1711337.0, "completions/mean_length": 124.875, "completions/min_length": 68.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 69.5714340209961, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9173502922058105, "rewards/meter/std": 0.16838125884532928, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8547643423080444, "rewards/total_composite/std": 0.34538865089416504, "reward": 0.8547643423080444, "reward_std": 0.34538865089416504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009888042695820332, "sampling/sampling_logp_difference/max": 0.9406858086585999, "sampling/importance_sampling_ratio/min": 0.390360027551651, "sampling/importance_sampling_ratio/mean": 1.003798246383667, "sampling/importance_sampling_ratio/max": 1.9665493965148926, "entropy": 0.027189649874344468, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.008955941535532475, "clip_ratio/high_max": 0.008955941535532475, "clip_ratio/region_mean": 0.008955941535532475, "reward_total_mean": 0.8547643423080444, "reward_meter_mean": 0.9173502922058105, "reward_meter_std": 0.16838125884532928, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8547643423080444, "reward_total_composite_std": 0.34538865089416504, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 789.0} {"timestamp_utc": "2026-04-11T21:04:59Z", "mode": "train", "global_step": 790, "epoch": 0.030506641952425084, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.609090909090909e-06, "num_tokens": 1713297.0, "completions/mean_length": 72.0, "completions/min_length": 72.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9889621734619141, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9889621734619141, "rewards/total_composite/std": 0.0, "reward": 0.9889621734619141, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0008448630687780678, "sampling/sampling_logp_difference/max": 0.017775291576981544, "sampling/importance_sampling_ratio/min": 0.998410701751709, "sampling/importance_sampling_ratio/mean": 1.0008271932601929, "sampling/importance_sampling_ratio/max": 1.0179342031478882, "entropy": 0.007650263491086662, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9889621734619141, "reward_meter_mean": 0.9889621734619141, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9889621734619141, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 790.0} {"timestamp_utc": "2026-04-11T21:05:04Z", "mode": "train", "global_step": 791, "epoch": 0.03054525795489651, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.606060606060606e-06, "num_tokens": 1715113.0, "completions/mean_length": 44.0, "completions/min_length": 44.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.0, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9859310984611511, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9859310984611511, "rewards/total_composite/std": 0.0, "reward": 0.9859310984611511, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0008520528208464384, "sampling/sampling_logp_difference/max": 0.022808797657489777, "sampling/importance_sampling_ratio/min": 0.977449357509613, "sampling/importance_sampling_ratio/mean": 1.0005165338516235, "sampling/importance_sampling_ratio/max": 1.0160062313079834, "entropy": 0.007817606383468956, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9859310984611511, "reward_meter_mean": 0.9859310984611511, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9859310984611511, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 791.0} {"timestamp_utc": "2026-04-11T21:05:10Z", "mode": "train", "global_step": 792, "epoch": 0.030583873957367932, "loss": -0.0222, "grad_norm": 4.220223426818848, "learning_rate": 7.603030303030303e-06, "num_tokens": 1717320.0, "completions/mean_length": 105.875, "completions/min_length": 98.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.875, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.8832316994667053, "rewards/meter/std": 0.2419254332780838, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8832316994667053, "rewards/total_composite/std": 0.2419254332780838, "reward": 0.8832316994667053, "reward_std": 0.2419254332780838, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024423938244581223, "sampling/sampling_logp_difference/max": 3.278282403945923, "sampling/importance_sampling_ratio/min": 0.03769294172525406, "sampling/importance_sampling_ratio/mean": 1.0003644227981567, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08906776923686266, "clip_ratio/low_mean": 0.008675742661580443, "clip_ratio/low_min": 0.008675742661580443, "clip_ratio/high_mean": 0.011705493438057601, "clip_ratio/high_max": 0.011705493438057601, "clip_ratio/region_mean": 0.020381236099638045, "reward_total_mean": 0.8832316994667053, "reward_meter_mean": 0.8832316994667053, "reward_meter_std": 0.2419254332780838, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8832316994667053, "reward_total_composite_std": 0.2419254332780838, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 792.0} {"timestamp_utc": "2026-04-11T21:05:14Z", "mode": "train", "global_step": 793, "epoch": 0.030622489959839357, "loss": 0.0078, "grad_norm": 2.699869394302368, "learning_rate": 7.600000000000001e-06, "num_tokens": 1719111.0, "completions/mean_length": 69.875, "completions/min_length": 69.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.875, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9781142473220825, "rewards/meter/std": 0.0427652932703495, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9781142473220825, "rewards/total_composite/std": 0.0427652932703495, "reward": 0.9781142473220825, "reward_std": 0.04276527464389801, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014242011122405529, "sampling/sampling_logp_difference/max": 1.3154139518737793, "sampling/importance_sampling_ratio/min": 0.571818470954895, "sampling/importance_sampling_ratio/mean": 1.0026626586914062, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05990207474678755, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/high_mean": 0.010740165715105832, "clip_ratio/high_max": 0.010740165715105832, "clip_ratio/region_mean": 0.012525880010798573, "reward_total_mean": 0.9781142473220825, "reward_meter_mean": 0.9781142473220825, "reward_meter_std": 0.0427652932703495, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9781142473220825, "reward_total_composite_std": 0.0427652932703495, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 793.0} {"timestamp_utc": "2026-04-11T21:05:20Z", "mode": "train", "global_step": 794, "epoch": 0.03066110596231078, "loss": 0.0198, "grad_norm": 1.6599212884902954, "learning_rate": 7.596969696969697e-06, "num_tokens": 1721331.0, "completions/mean_length": 112.5, "completions/min_length": 111.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.5, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.87825608253479, "rewards/meter/std": 0.33554860949516296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.87825608253479, "rewards/total_composite/std": 0.33554860949516296, "reward": 0.87825608253479, "reward_std": 0.33554860949516296, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011948403902351856, "sampling/sampling_logp_difference/max": 1.4520015716552734, "sampling/importance_sampling_ratio/min": 0.23410125076770782, "sampling/importance_sampling_ratio/mean": 0.9973942041397095, "sampling/importance_sampling_ratio/max": 1.5048034191131592, "entropy": 0.03857563203200698, "clip_ratio/low_mean": 0.0021186440717428923, "clip_ratio/low_min": 0.0021186440717428923, "clip_ratio/high_mean": 0.01228755246847868, "clip_ratio/high_max": 0.01228755246847868, "clip_ratio/region_mean": 0.014406196540221572, "reward_total_mean": 0.87825608253479, "reward_meter_mean": 0.87825608253479, "reward_meter_std": 0.33554860949516296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.87825608253479, "reward_total_composite_std": 0.33554860949516296, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 794.0} {"timestamp_utc": "2026-04-11T21:05:30Z", "mode": "train", "global_step": 795, "epoch": 0.030699721964782205, "loss": -0.1199, "grad_norm": 3.6339054107666016, "learning_rate": 7.593939393939395e-06, "num_tokens": 1723130.0, "completions/mean_length": 112.875, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 55.857147216796875, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8027093410491943, "rewards/meter/std": 0.3521903157234192, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7851887941360474, "rewards/total_composite/std": 0.3911862075328827, "reward": 0.7851887941360474, "reward_std": 0.3911861777305603, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09941212087869644, "sampling/sampling_logp_difference/max": 4.045551776885986, "sampling/importance_sampling_ratio/min": 0.017500044777989388, "sampling/importance_sampling_ratio/mean": 0.9849594235420227, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13437956385314465, "clip_ratio/low_mean": 0.004545454401522875, "clip_ratio/low_min": 0.004545454401522875, "clip_ratio/high_mean": 0.049006874207407236, "clip_ratio/high_max": 0.049006874207407236, "clip_ratio/region_mean": 0.05355232860893011, "reward_total_mean": 0.7851887941360474, "reward_meter_mean": 0.8027093410491943, "reward_meter_std": 0.3521903157234192, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7851887941360474, "reward_total_composite_std": 0.3911862075328827, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 795.0} {"timestamp_utc": "2026-04-11T21:05:36Z", "mode": "train", "global_step": 796, "epoch": 0.03073833796725363, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.590909090909091e-06, "num_tokens": 1726050.0, "completions/mean_length": 165.0, "completions/min_length": 161.0, "completions/max_length": 177.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 165.0, "completions/min_terminated_length": 161.0, "completions/max_terminated_length": 177.0, "rewards/meter/mean": 0.9958475828170776, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9958475828170776, "rewards/total_composite/std": 0.0, "reward": 0.9958475828170776, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.005931791849434376, "sampling/sampling_logp_difference/max": 3.6604130268096924, "sampling/importance_sampling_ratio/min": 0.02572188712656498, "sampling/importance_sampling_ratio/mean": 0.997971773147583, "sampling/importance_sampling_ratio/max": 1.1137677431106567, "entropy": 0.0030262921354733407, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9958475828170776, "reward_meter_mean": 0.9958475828170776, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9958475828170776, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 796.0} {"timestamp_utc": "2026-04-11T21:05:42Z", "mode": "train", "global_step": 797, "epoch": 0.030776953969725053, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.587878787878788e-06, "num_tokens": 1728306.0, "completions/mean_length": 88.0, "completions/min_length": 88.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.0, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9883676171302795, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9883676171302795, "rewards/total_composite/std": 0.0, "reward": 0.9883676171302795, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002712436835281551, "sampling/sampling_logp_difference/max": 0.00954503659158945, "sampling/importance_sampling_ratio/min": 0.9905003905296326, "sampling/importance_sampling_ratio/mean": 1.0001798868179321, "sampling/importance_sampling_ratio/max": 1.0093083381652832, "entropy": 0.0021069108188385144, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9883676171302795, "reward_meter_mean": 0.9883676171302795, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9883676171302795, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 797.0} {"timestamp_utc": "2026-04-11T21:05:46Z", "mode": "train", "global_step": 798, "epoch": 0.030815569972196477, "loss": 0.0029, "grad_norm": 1.0194988250732422, "learning_rate": 7.584848484848486e-06, "num_tokens": 1729986.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9984152317047119, "rewards/meter/std": 0.00010643188579706475, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984152317047119, "rewards/total_composite/std": 0.00010643188579706475, "reward": 0.9984152317047119, "reward_std": 0.00010642191773513332, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009082009084522724, "sampling/sampling_logp_difference/max": 0.8598246574401855, "sampling/importance_sampling_ratio/min": 0.42323628067970276, "sampling/importance_sampling_ratio/mean": 1.001835584640503, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0273487598169595, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/high_mean": 0.008196720853447914, "clip_ratio/high_max": 0.008196720853447914, "clip_ratio/region_mean": 0.012295081280171871, "reward_total_mean": 0.9984152317047119, "reward_meter_mean": 0.9984152317047119, "reward_meter_std": 0.00010643188579706475, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984152317047119, "reward_total_composite_std": 0.00010643188579706475, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 798.0} {"timestamp_utc": "2026-04-11T21:05:56Z", "mode": "train", "global_step": 799, "epoch": 0.0308541859746679, "loss": -0.0734, "grad_norm": 0.9545469284057617, "learning_rate": 7.581818181818183e-06, "num_tokens": 1731599.0, "completions/mean_length": 232.625, "completions/min_length": 59.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.07764222472906113, "rewards/meter/std": 0.05144652724266052, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.06288698315620422, "rewards/total_composite/std": 0.04681145399808884, "reward": 0.06288698315620422, "reward_std": 0.04681145399808884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.052948325872421265, "sampling/sampling_logp_difference/max": 3.1623458862304688, "sampling/importance_sampling_ratio/min": 0.04232633113861084, "sampling/importance_sampling_ratio/mean": 1.0014253854751587, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1024195933714509, "clip_ratio/low_mean": 0.004166666883975267, "clip_ratio/low_min": 0.004166666883975267, "clip_ratio/high_mean": 0.017176149878650904, "clip_ratio/high_max": 0.017176149878650904, "clip_ratio/region_mean": 0.02134281676262617, "reward_total_mean": 0.06288698315620422, "reward_meter_mean": 0.07764222472906113, "reward_meter_std": 0.05144652724266052, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.06288698315620422, "reward_total_composite_std": 0.04681145399808884, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 799.0} {"timestamp_utc": "2026-04-11T21:06:01Z", "mode": "train", "global_step": 800, "epoch": 0.030892801977139325, "loss": 0.0496, "grad_norm": 7.082140922546387, "learning_rate": 7.57878787878788e-06, "num_tokens": 1733574.0, "completions/mean_length": 75.875, "completions/min_length": 71.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.875, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.4755059778690338, "rewards/meter/std": 0.411925345659256, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4755059778690338, "rewards/total_composite/std": 0.411925345659256, "reward": 0.4755059778690338, "reward_std": 0.41192537546157837, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030456366017460823, "sampling/sampling_logp_difference/max": 4.81218147277832, "sampling/importance_sampling_ratio/min": 0.008130105212330818, "sampling/importance_sampling_ratio/mean": 0.99607914686203, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0573624970857054, "clip_ratio/low_mean": 0.016302418429404497, "clip_ratio/low_min": 0.016302418429404497, "clip_ratio/high_mean": 0.011967072496190667, "clip_ratio/high_max": 0.011967072496190667, "clip_ratio/region_mean": 0.028269490925595164, "reward_total_mean": 0.4755059778690338, "reward_meter_mean": 0.4755059778690338, "reward_meter_std": 0.411925345659256, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4755059778690338, "reward_total_composite_std": 0.411925345659256, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 800.0} {"timestamp_utc": "2026-04-11T21:07:27Z", "mode": "eval", "global_step": 800, "epoch": 0.030892801977139325, "eval_loss": NaN, "eval_runtime": 85.4249, "eval_samples_per_second": 1.217, "eval_steps_per_second": 0.152, "eval_num_tokens": 1733574.0, "eval_completions/mean_length": 215.28846153846155, "eval_completions/min_length": 53.92307692307692, "eval_completions/max_length": 460.9230769230769, "eval_completions/clipped_ratio": 0.08653846153846154, "eval_completions/mean_terminated_length": 187.47482299804688, "eval_completions/min_terminated_length": 53.92307692307692, "eval_completions/max_terminated_length": 378.15384615384613, "eval_rewards/meter/mean": 0.6363248893847833, "eval_rewards/meter/std": 0.42047716276003766, "eval_rewards/count_adherence/mean": 0.9369661578765283, "eval_rewards/count_adherence/std": 0.11172813578293873, "eval_rewards/arabic_clean/mean": 0.9423076923076923, "eval_rewards/arabic_clean/std": 0.1443941226372352, "eval_rewards/total_composite/mean": 0.600143565581395, "eval_rewards/total_composite/std": 0.4244091361761093, "eval_reward": 0.600143565581395, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.004912831847412655, "eval_sampling/sampling_logp_difference/max": 0.6944233820988581, "eval_sampling/importance_sampling_ratio/min": 0.5719292301398057, "eval_sampling/importance_sampling_ratio/mean": 1.0011764031190138, "eval_sampling/importance_sampling_ratio/max": 1.3467054275365977, "eval_entropy": 0.043126778748746104, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.600143565581395, "eval_reward_meter_mean": 0.6363248893847833, "eval_reward_meter_std": 0.42047716276003766, "eval_reward_count_adherence_mean": 0.9369661578765283, "eval_reward_count_adherence_std": 0.11172813578293873, "eval_reward_arabic_clean_mean": 0.9423076923076923, "eval_reward_arabic_clean_std": 0.1443941226372352, "eval_reward_total_composite_mean": 0.600143565581395, "eval_reward_total_composite_std": 0.4244091361761093, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 800.0} {"timestamp_utc": "2026-04-11T21:07:34Z", "mode": "train", "global_step": 801, "epoch": 0.03093141797961075, "loss": 0.0142, "grad_norm": 4.123575210571289, "learning_rate": 7.5757575757575764e-06, "num_tokens": 1735372.0, "completions/mean_length": 69.75, "completions/min_length": 64.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.6018670797348022, "rewards/meter/std": 0.41990745067596436, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6018670797348022, "rewards/total_composite/std": 0.41990745067596436, "reward": 0.6018670797348022, "reward_std": 0.41990742087364197, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023207932710647583, "sampling/sampling_logp_difference/max": 0.841026782989502, "sampling/importance_sampling_ratio/min": 0.4947810471057892, "sampling/importance_sampling_ratio/mean": 1.0090745687484741, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1263660485856235, "clip_ratio/low_mean": 0.008892111829482019, "clip_ratio/low_min": 0.008892111829482019, "clip_ratio/high_mean": 0.007336147828027606, "clip_ratio/high_max": 0.007336147828027606, "clip_ratio/region_mean": 0.016228259657509625, "reward_total_mean": 0.6018670797348022, "reward_meter_mean": 0.6018670797348022, "reward_meter_std": 0.41990745067596436, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6018670797348022, "reward_total_composite_std": 0.41990745067596436, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 801.0} {"timestamp_utc": "2026-04-11T21:07:40Z", "mode": "train", "global_step": 802, "epoch": 0.030970033982082174, "loss": -0.0066, "grad_norm": 1.9443621635437012, "learning_rate": 7.572727272727274e-06, "num_tokens": 1737781.0, "completions/mean_length": 125.125, "completions/min_length": 123.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.125, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.99431312084198, "rewards/meter/std": 0.00208661169745028, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.99431312084198, "rewards/total_composite/std": 0.00208661169745028, "reward": 0.99431312084198, "reward_std": 0.0020866054110229015, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012328991666436195, "sampling/sampling_logp_difference/max": 1.3417401313781738, "sampling/importance_sampling_ratio/min": 0.26139041781425476, "sampling/importance_sampling_ratio/mean": 1.000008225440979, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.043647154700011015, "clip_ratio/low_mean": 0.005081300623714924, "clip_ratio/low_min": 0.005081300623714924, "clip_ratio/high_mean": 0.004970373585820198, "clip_ratio/high_max": 0.004970373585820198, "clip_ratio/region_mean": 0.010051674209535122, "reward_total_mean": 0.99431312084198, "reward_meter_mean": 0.99431312084198, "reward_meter_std": 0.00208661169745028, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.99431312084198, "reward_total_composite_std": 0.00208661169745028, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 802.0} {"timestamp_utc": "2026-04-11T21:07:45Z", "mode": "train", "global_step": 803, "epoch": 0.031008649984553598, "loss": 0.0162, "grad_norm": 2.924515962600708, "learning_rate": 7.56969696969697e-06, "num_tokens": 1739875.0, "completions/mean_length": 94.75, "completions/min_length": 93.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.75, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9924517869949341, "rewards/meter/std": 0.007060833275318146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9924517869949341, "rewards/total_composite/std": 0.007060833275318146, "reward": 0.9924517869949341, "reward_std": 0.007060818839818239, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010468529537320137, "sampling/sampling_logp_difference/max": 1.654249906539917, "sampling/importance_sampling_ratio/min": 0.19123543798923492, "sampling/importance_sampling_ratio/mean": 1.0004314184188843, "sampling/importance_sampling_ratio/max": 1.2555328607559204, "entropy": 0.04652517521753907, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0026598755503073335, "clip_ratio/high_max": 0.0026598755503073335, "clip_ratio/region_mean": 0.0026598755503073335, "reward_total_mean": 0.9924517869949341, "reward_meter_mean": 0.9924517869949341, "reward_meter_std": 0.007060833275318146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9924517869949341, "reward_total_composite_std": 0.007060833275318146, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 803.0} {"timestamp_utc": "2026-04-11T21:07:55Z", "mode": "train", "global_step": 804, "epoch": 0.03104726598702502, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.566666666666667e-06, "num_tokens": 1745627.0, "completions/mean_length": 481.0, "completions/min_length": 481.0, "completions/max_length": 481.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 481.0, "completions/min_terminated_length": 481.0, "completions/max_terminated_length": 481.0, "rewards/meter/mean": 0.9958475828170776, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9336071014404297, "rewards/total_composite/std": 0.0, "reward": 0.9336071014404297, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.000563071807846427, "sampling/sampling_logp_difference/max": 1.2687495946884155, "sampling/importance_sampling_ratio/min": 0.28118300437927246, "sampling/importance_sampling_ratio/mean": 0.999758780002594, "sampling/importance_sampling_ratio/max": 1.0688825845718384, "entropy": 0.0015184664662228897, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9336071014404297, "reward_meter_mean": 0.9958475828170776, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9336071014404297, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 804.0} {"timestamp_utc": "2026-04-11T21:08:00Z", "mode": "train", "global_step": 805, "epoch": 0.031085881989496446, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.563636363636364e-06, "num_tokens": 1747835.0, "completions/mean_length": 98.0, "completions/min_length": 98.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9957982897758484, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9957982897758484, "rewards/total_composite/std": 0.0, "reward": 0.9957982897758484, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.000553442572709173, "sampling/sampling_logp_difference/max": 0.045761480927467346, "sampling/importance_sampling_ratio/min": 0.9989988207817078, "sampling/importance_sampling_ratio/mean": 1.0005515813827515, "sampling/importance_sampling_ratio/max": 1.0468246936798096, "entropy": 0.005296186718624085, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9957982897758484, "reward_meter_mean": 0.9957982897758484, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9957982897758484, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 805.0} {"timestamp_utc": "2026-04-11T21:08:10Z", "mode": "train", "global_step": 806, "epoch": 0.03112449799196787, "loss": -0.2389, "grad_norm": 0.5651002526283264, "learning_rate": 7.560606060606062e-06, "num_tokens": 1750383.0, "completions/mean_length": 198.5, "completions/min_length": 148.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 153.71429443359375, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.8478649854660034, "rewards/meter/std": 0.3413846492767334, "rewards/count_adherence/mean": 0.8958333134651184, "rewards/count_adherence/std": 0.294627845287323, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8474308252334595, "rewards/total_composite/std": 0.342612087726593, "reward": 0.8474308252334595, "reward_std": 0.342612087726593, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01576400361955166, "sampling/sampling_logp_difference/max": 1.948196291923523, "sampling/importance_sampling_ratio/min": 0.1425309181213379, "sampling/importance_sampling_ratio/mean": 0.997974693775177, "sampling/importance_sampling_ratio/max": 1.9672658443450928, "entropy": 0.042604236863553524, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.013878182100597769, "clip_ratio/high_max": 0.013878182100597769, "clip_ratio/region_mean": 0.013878182100597769, "reward_total_mean": 0.8474308252334595, "reward_meter_mean": 0.8478649854660034, "reward_meter_std": 0.3413846492767334, "reward_count_adherence_mean": 0.8958333134651184, "reward_count_adherence_std": 0.294627845287323, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8474308252334595, "reward_total_composite_std": 0.342612087726593, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 806.0} {"timestamp_utc": "2026-04-11T21:08:15Z", "mode": "train", "global_step": 807, "epoch": 0.031163113994439294, "loss": 0.0204, "grad_norm": 3.7735772132873535, "learning_rate": 7.557575757575758e-06, "num_tokens": 1752200.0, "completions/mean_length": 61.125, "completions/min_length": 60.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.970656156539917, "rewards/meter/std": 0.06519167870283127, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.970656156539917, "rewards/total_composite/std": 0.06519167870283127, "reward": 0.970656156539917, "reward_std": 0.06519167125225067, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029046399518847466, "sampling/sampling_logp_difference/max": 1.0587835311889648, "sampling/importance_sampling_ratio/min": 0.3468775153160095, "sampling/importance_sampling_ratio/mean": 1.008776307106018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13918291218578815, "clip_ratio/low_mean": 0.005859375, "clip_ratio/low_min": 0.005859375, "clip_ratio/high_mean": 0.024627622216939926, "clip_ratio/high_max": 0.024627622216939926, "clip_ratio/region_mean": 0.030486997216939926, "reward_total_mean": 0.970656156539917, "reward_meter_mean": 0.970656156539917, "reward_meter_std": 0.06519167870283127, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.970656156539917, "reward_total_composite_std": 0.06519167870283127, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 807.0} {"timestamp_utc": "2026-04-11T21:08:19Z", "mode": "train", "global_step": 808, "epoch": 0.031201729996910718, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.5545454545454555e-06, "num_tokens": 1753704.0, "completions/mean_length": 50.0, "completions/min_length": 50.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9953853487968445, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9953853487968445, "rewards/total_composite/std": 0.0, "reward": 0.9953853487968445, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0011840584920719266, "sampling/sampling_logp_difference/max": 0.05418664216995239, "sampling/importance_sampling_ratio/min": 0.9951973557472229, "sampling/importance_sampling_ratio/mean": 1.0011436939239502, "sampling/importance_sampling_ratio/max": 1.055681586265564, "entropy": 0.008669113391079009, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9953853487968445, "reward_meter_mean": 0.9953853487968445, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9953853487968445, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 808.0} {"timestamp_utc": "2026-04-11T21:08:29Z", "mode": "train", "global_step": 809, "epoch": 0.031240345999382142, "loss": -0.0605, "grad_norm": 2.7631633281707764, "learning_rate": 7.551515151515152e-06, "num_tokens": 1755731.0, "completions/mean_length": 148.375, "completions/min_length": 88.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 96.42857360839844, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.830810546875, "rewards/meter/std": 0.227777361869812, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7917212843894958, "rewards/total_composite/std": 0.23348788917064667, "reward": 0.7917212843894958, "reward_std": 0.23348785936832428, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03167222812771797, "sampling/sampling_logp_difference/max": 1.5958867073059082, "sampling/importance_sampling_ratio/min": 0.20272868871688843, "sampling/importance_sampling_ratio/mean": 0.9961344599723816, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11443378031253815, "clip_ratio/low_mean": 0.013008257374167442, "clip_ratio/low_min": 0.013008257374167442, "clip_ratio/high_mean": 0.01031811349093914, "clip_ratio/high_max": 0.01031811349093914, "clip_ratio/region_mean": 0.023326370865106583, "reward_total_mean": 0.7917212843894958, "reward_meter_mean": 0.830810546875, "reward_meter_std": 0.227777361869812, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7917212843894958, "reward_total_composite_std": 0.23348788917064667, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 809.0} {"timestamp_utc": "2026-04-11T21:08:39Z", "mode": "train", "global_step": 810, "epoch": 0.031278962001853566, "loss": -0.1832, "grad_norm": 0.9950336813926697, "learning_rate": 7.548484848484849e-06, "num_tokens": 1757741.0, "completions/mean_length": 144.25, "completions/min_length": 89.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 91.71428680419922, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.8862010836601257, "rewards/meter/std": 0.153972327709198, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8095306754112244, "rewards/total_composite/std": 0.3443083167076111, "reward": 0.8095306754112244, "reward_std": 0.3443083167076111, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020651884377002716, "sampling/sampling_logp_difference/max": 0.9706335067749023, "sampling/importance_sampling_ratio/min": 0.37884294986724854, "sampling/importance_sampling_ratio/mean": 1.0004615783691406, "sampling/importance_sampling_ratio/max": 1.777065634727478, "entropy": 0.09916476998478174, "clip_ratio/low_mean": 0.002659574383869767, "clip_ratio/low_min": 0.002659574383869767, "clip_ratio/high_mean": 0.015102508361451328, "clip_ratio/high_max": 0.015102508361451328, "clip_ratio/region_mean": 0.017762082745321095, "reward_total_mean": 0.8095306754112244, "reward_meter_mean": 0.8862010836601257, "reward_meter_std": 0.153972327709198, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8095306754112244, "reward_total_composite_std": 0.3443083167076111, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 810.0} {"timestamp_utc": "2026-04-11T21:08:49Z", "mode": "train", "global_step": 811, "epoch": 0.03131757800432499, "loss": -0.0497, "grad_norm": 1.1814782619476318, "learning_rate": 7.545454545454546e-06, "num_tokens": 1759168.0, "completions/mean_length": 400.375, "completions/min_length": 64.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.75, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.0093841552734375, "rewards/meter/std": 0.011478016152977943, "rewards/count_adherence/mean": 0.5625, "rewards/count_adherence/std": 0.4172614812850952, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/total_composite/mean": 0.004483422264456749, "rewards/total_composite/std": 0.006689653266221285, "reward": 0.004483422264456749, "reward_std": 0.0066896528005599976, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17178411781787872, "sampling/sampling_logp_difference/max": 5.017492771148682, "sampling/importance_sampling_ratio/min": 0.006621106527745724, "sampling/importance_sampling_ratio/mean": 0.9806210994720459, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09933926910161972, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.023000232875347137, "clip_ratio/high_max": 0.023000232875347137, "clip_ratio/region_mean": 0.023000232875347137, "reward_total_mean": 0.004483422264456749, "reward_meter_mean": 0.0093841552734375, "reward_meter_std": 0.011478016152977943, "reward_count_adherence_mean": 0.5625, "reward_count_adherence_std": 0.4172614812850952, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_total_composite_mean": 0.004483422264456749, "reward_total_composite_std": 0.006689653266221285, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 811.0} {"timestamp_utc": "2026-04-11T21:08:54Z", "mode": "train", "global_step": 812, "epoch": 0.031356194006796415, "loss": -0.0077, "grad_norm": 1.2783820629119873, "learning_rate": 7.542424242424244e-06, "num_tokens": 1761454.0, "completions/mean_length": 117.75, "completions/min_length": 116.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.75, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9915919303894043, "rewards/meter/std": 0.0008497779490426183, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9915919303894043, "rewards/total_composite/std": 0.0008497779490426183, "reward": 0.9915919303894043, "reward_std": 0.0008497649105265737, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012695571407675743, "sampling/sampling_logp_difference/max": 1.5787792205810547, "sampling/importance_sampling_ratio/min": 0.20622670650482178, "sampling/importance_sampling_ratio/mean": 0.9987900257110596, "sampling/importance_sampling_ratio/max": 1.867525339126587, "entropy": 0.05143675860017538, "clip_ratio/low_mean": 0.005360300652682781, "clip_ratio/low_min": 0.005360300652682781, "clip_ratio/high_mean": 0.006235800567083061, "clip_ratio/high_max": 0.006235800567083061, "clip_ratio/region_mean": 0.011596101219765842, "reward_total_mean": 0.9915919303894043, "reward_meter_mean": 0.9915919303894043, "reward_meter_std": 0.0008497779490426183, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9915919303894043, "reward_total_composite_std": 0.0008497779490426183, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 812.0} {"timestamp_utc": "2026-04-11T21:08:59Z", "mode": "train", "global_step": 813, "epoch": 0.03139481000926784, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.53939393939394e-06, "num_tokens": 1763310.0, "completions/mean_length": 74.0, "completions/min_length": 74.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.0, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9956606030464172, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9956606030464172, "rewards/total_composite/std": 0.0, "reward": 0.9956606030464172, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0007216231315396726, "sampling/sampling_logp_difference/max": 0.042141228914260864, "sampling/importance_sampling_ratio/min": 0.9981146454811096, "sampling/importance_sampling_ratio/mean": 1.0007139444351196, "sampling/importance_sampling_ratio/max": 1.0430418252944946, "entropy": 0.005994105304125696, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9956606030464172, "reward_meter_mean": 0.9956606030464172, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9956606030464172, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 813.0} {"timestamp_utc": "2026-04-11T21:09:09Z", "mode": "train", "global_step": 814, "epoch": 0.03143342601173926, "loss": -0.0585, "grad_norm": 1.601417064666748, "learning_rate": 7.536363636363637e-06, "num_tokens": 1764701.0, "completions/mean_length": 213.875, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.4680856466293335, "rewards/meter/std": 0.3238549828529358, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.40284979343414307, "rewards/total_composite/std": 0.39089274406433105, "reward": 0.40284979343414307, "reward_std": 0.39089271426200867, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08656477928161621, "sampling/sampling_logp_difference/max": 2.7611823081970215, "sampling/importance_sampling_ratio/min": 0.06321697682142258, "sampling/importance_sampling_ratio/mean": 1.0082935094833374, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2470087818801403, "clip_ratio/low_mean": 0.013513513840734959, "clip_ratio/low_min": 0.013513513840734959, "clip_ratio/high_mean": 0.03962418343871832, "clip_ratio/high_max": 0.03962418343871832, "clip_ratio/region_mean": 0.05313769727945328, "reward_total_mean": 0.40284979343414307, "reward_meter_mean": 0.4680856466293335, "reward_meter_std": 0.3238549828529358, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.40284979343414307, "reward_total_composite_std": 0.39089274406433105, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 814.0} {"timestamp_utc": "2026-04-11T21:09:14Z", "mode": "train", "global_step": 815, "epoch": 0.03147204201421069, "loss": 0.0735, "grad_norm": 27.867839813232422, "learning_rate": 7.533333333333334e-06, "num_tokens": 1766362.0, "completions/mean_length": 37.625, "completions/min_length": 35.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.5993378162384033, "rewards/meter/std": 0.27688512206077576, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5993378162384033, "rewards/total_composite/std": 0.27688512206077576, "reward": 0.5993378162384033, "reward_std": 0.27688512206077576, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09992975741624832, "sampling/sampling_logp_difference/max": 4.163671493530273, "sampling/importance_sampling_ratio/min": 0.015550361014902592, "sampling/importance_sampling_ratio/mean": 0.9893629550933838, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15575411915779114, "clip_ratio/low_mean": 0.01222420996055007, "clip_ratio/low_min": 0.01222420996055007, "clip_ratio/high_mean": 0.021230158861726522, "clip_ratio/high_max": 0.021230158861726522, "clip_ratio/region_mean": 0.03345436882227659, "reward_total_mean": 0.5993378162384033, "reward_meter_mean": 0.5993378162384033, "reward_meter_std": 0.27688512206077576, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5993378162384033, "reward_total_composite_std": 0.27688512206077576, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 815.0} {"timestamp_utc": "2026-04-11T21:09:19Z", "mode": "train", "global_step": 816, "epoch": 0.03151065801668211, "loss": 0.0316, "grad_norm": 4.1365532875061035, "learning_rate": 7.530303030303031e-06, "num_tokens": 1768191.0, "completions/mean_length": 58.625, "completions/min_length": 51.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.4646347761154175, "rewards/meter/std": 0.18640866875648499, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4646347761154175, "rewards/total_composite/std": 0.18640866875648499, "reward": 0.4646347761154175, "reward_std": 0.18640868365764618, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02432762272655964, "sampling/sampling_logp_difference/max": 1.1967215538024902, "sampling/importance_sampling_ratio/min": 0.30218327045440674, "sampling/importance_sampling_ratio/mean": 1.0047223567962646, "sampling/importance_sampling_ratio/max": 1.8075611591339111, "entropy": 0.09666818100959063, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.010390537790954113, "clip_ratio/high_max": 0.010390537790954113, "clip_ratio/region_mean": 0.014296787790954113, "reward_total_mean": 0.4646347761154175, "reward_meter_mean": 0.4646347761154175, "reward_meter_std": 0.18640866875648499, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4646347761154175, "reward_total_composite_std": 0.18640866875648499, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 816.0} {"timestamp_utc": "2026-04-11T21:09:28Z", "mode": "train", "global_step": 817, "epoch": 0.031549274019153535, "loss": 0.4025, "grad_norm": 4.435557842254639, "learning_rate": 7.5272727272727274e-06, "num_tokens": 1770536.0, "completions/mean_length": 125.125, "completions/min_length": 75.0, "completions/max_length": 411.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.125, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 411.0, "rewards/meter/mean": 0.43105483055114746, "rewards/meter/std": 0.290896475315094, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.3998875021934509, "rewards/total_composite/std": 0.32455718517303467, "reward": 0.3998875021934509, "reward_std": 0.32455718517303467, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08144014328718185, "sampling/sampling_logp_difference/max": 1.5423645973205566, "sampling/importance_sampling_ratio/min": 0.21387477219104767, "sampling/importance_sampling_ratio/mean": 1.0054211616516113, "sampling/importance_sampling_ratio/max": 1.9394710063934326, "entropy": 0.964023835491389, "clip_ratio/low_mean": 0.013261043233796954, "clip_ratio/low_min": 0.013261043233796954, "clip_ratio/high_mean": 0.012935383478179574, "clip_ratio/high_max": 0.012935383478179574, "clip_ratio/region_mean": 0.026196426711976528, "reward_total_mean": 0.3998875021934509, "reward_meter_mean": 0.43105483055114746, "reward_meter_std": 0.290896475315094, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.3998875021934509, "reward_total_composite_std": 0.32455718517303467, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 817.0} {"timestamp_utc": "2026-04-11T21:09:38Z", "mode": "train", "global_step": 818, "epoch": 0.03158789002162496, "loss": -0.0059, "grad_norm": 0.35244715213775635, "learning_rate": 7.524242424242425e-06, "num_tokens": 1772283.0, "completions/mean_length": 435.375, "completions/min_length": 62.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.75, "completions/mean_terminated_length": 205.5, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 349.0, "rewards/meter/mean": 0.6548354625701904, "rewards/meter/std": 0.43205440044403076, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.25877460837364197, "rewards/arabic_clean/mean": 0.125, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.1189679205417633, "rewards/total_composite/std": 0.3364920914173126, "reward": 0.1189679205417633, "reward_std": 0.3364920914173126, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13183903694152832, "sampling/sampling_logp_difference/max": 1.1653553247451782, "sampling/importance_sampling_ratio/min": 0.3118118643760681, "sampling/importance_sampling_ratio/mean": 1.0293099880218506, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8205861123278737, "clip_ratio/low_mean": 0.003223495790734887, "clip_ratio/low_min": 0.003223495790734887, "clip_ratio/high_mean": 0.004032257944345474, "clip_ratio/high_max": 0.004032257944345474, "clip_ratio/region_mean": 0.007255753735080361, "reward_total_mean": 0.1189679205417633, "reward_meter_mean": 0.6548354625701904, "reward_meter_std": 0.43205440044403076, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.25877460837364197, "reward_arabic_clean_mean": 0.125, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.1189679205417633, "reward_total_composite_std": 0.3364920914173126, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 818.0} {"timestamp_utc": "2026-04-11T21:09:43Z", "mode": "train", "global_step": 819, "epoch": 0.03162650602409638, "loss": 0.0438, "grad_norm": 5.961421489715576, "learning_rate": 7.521212121212121e-06, "num_tokens": 1774235.0, "completions/mean_length": 76.0, "completions/min_length": 73.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.6910608410835266, "rewards/meter/std": 0.33199170231819153, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6910608410835266, "rewards/total_composite/std": 0.33199170231819153, "reward": 0.6910608410835266, "reward_std": 0.33199167251586914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05064326897263527, "sampling/sampling_logp_difference/max": 12.507050514221191, "sampling/importance_sampling_ratio/min": 3.7004706427978817e-06, "sampling/importance_sampling_ratio/mean": 0.9978348612785339, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06165915634483099, "clip_ratio/low_mean": 0.0189310887362808, "clip_ratio/low_min": 0.0189310887362808, "clip_ratio/high_mean": 0.0051369862630963326, "clip_ratio/high_max": 0.0051369862630963326, "clip_ratio/region_mean": 0.02406807499937713, "reward_total_mean": 0.6910608410835266, "reward_meter_mean": 0.6910608410835266, "reward_meter_std": 0.33199170231819153, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6910608410835266, "reward_total_composite_std": 0.33199170231819153, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 819.0} {"timestamp_utc": "2026-04-11T21:09:48Z", "mode": "train", "global_step": 820, "epoch": 0.03166512202656781, "loss": 0.0158, "grad_norm": 4.252890586853027, "learning_rate": 7.518181818181819e-06, "num_tokens": 1775922.0, "completions/mean_length": 58.875, "completions/min_length": 56.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9819810390472412, "rewards/meter/std": 0.007964253425598145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9819810390472412, "rewards/total_composite/std": 0.007964253425598145, "reward": 0.9819810390472412, "reward_std": 0.007964243181049824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031156206503510475, "sampling/sampling_logp_difference/max": 1.4427082538604736, "sampling/importance_sampling_ratio/min": 0.2362869828939438, "sampling/importance_sampling_ratio/mean": 0.9951539635658264, "sampling/importance_sampling_ratio/max": 1.6159226894378662, "entropy": 0.10328649077564478, "clip_ratio/low_mean": 0.03168201702646911, "clip_ratio/low_min": 0.03168201702646911, "clip_ratio/high_mean": 0.004350787028670311, "clip_ratio/high_max": 0.004350787028670311, "clip_ratio/region_mean": 0.03603280405513942, "reward_total_mean": 0.9819810390472412, "reward_meter_mean": 0.9819810390472412, "reward_meter_std": 0.007964253425598145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9819810390472412, "reward_total_composite_std": 0.007964253425598145, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 820.0} {"timestamp_utc": "2026-04-11T21:09:52Z", "mode": "train", "global_step": 821, "epoch": 0.03170373802903923, "loss": 0.0038, "grad_norm": 2.7222046852111816, "learning_rate": 7.515151515151516e-06, "num_tokens": 1777651.0, "completions/mean_length": 50.125, "completions/min_length": 50.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.125, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9954612255096436, "rewards/meter/std": 0.00021465388999786228, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9954612255096436, "rewards/total_composite/std": 0.00021465388999786228, "reward": 0.9954612255096436, "reward_std": 0.00021465991449076682, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00853498000651598, "sampling/sampling_logp_difference/max": 1.846993088722229, "sampling/importance_sampling_ratio/min": 0.15771067142486572, "sampling/importance_sampling_ratio/mean": 0.9974405765533447, "sampling/importance_sampling_ratio/max": 1.065722107887268, "entropy": 0.020875168032944202, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0049019609577953815, "clip_ratio/high_max": 0.0049019609577953815, "clip_ratio/region_mean": 0.0049019609577953815, "reward_total_mean": 0.9954612255096436, "reward_meter_mean": 0.9954612255096436, "reward_meter_std": 0.00021465388999786228, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9954612255096436, "reward_total_composite_std": 0.00021465388999786228, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 821.0} {"timestamp_utc": "2026-04-11T21:09:58Z", "mode": "train", "global_step": 822, "epoch": 0.031742354031510656, "loss": 0.0134, "grad_norm": 1.4621386528015137, "learning_rate": 7.512121212121213e-06, "num_tokens": 1780165.0, "completions/mean_length": 128.25, "completions/min_length": 123.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.25, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9953622817993164, "rewards/meter/std": 0.00053422711789608, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9953622817993164, "rewards/total_composite/std": 0.00053422711789608, "reward": 0.9953622817993164, "reward_std": 0.000534241960849613, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0032333482522517443, "sampling/sampling_logp_difference/max": 0.6709310412406921, "sampling/importance_sampling_ratio/min": 0.5112323760986328, "sampling/importance_sampling_ratio/mean": 0.9993246793746948, "sampling/importance_sampling_ratio/max": 1.1764084100723267, "entropy": 0.015514681115746498, "clip_ratio/low_mean": 0.0029069767333567142, "clip_ratio/low_min": 0.0029069767333567142, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0029069767333567142, "reward_total_mean": 0.9953622817993164, "reward_meter_mean": 0.9953622817993164, "reward_meter_std": 0.00053422711789608, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9953622817993164, "reward_total_composite_std": 0.00053422711789608, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 822.0} {"timestamp_utc": "2026-04-11T21:10:02Z", "mode": "train", "global_step": 823, "epoch": 0.03178097003398208, "loss": -0.0094, "grad_norm": 27.496055603027344, "learning_rate": 7.509090909090909e-06, "num_tokens": 1781931.0, "completions/mean_length": 43.75, "completions/min_length": 42.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.75, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9838736057281494, "rewards/meter/std": 0.0036353026516735554, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9838736057281494, "rewards/total_composite/std": 0.0036353026516735554, "reward": 0.9838736057281494, "reward_std": 0.003635299624875188, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009803482331335545, "sampling/sampling_logp_difference/max": 1.2487645149230957, "sampling/importance_sampling_ratio/min": 0.28685900568962097, "sampling/importance_sampling_ratio/mean": 0.999248743057251, "sampling/importance_sampling_ratio/max": 1.3416532278060913, "entropy": 0.028031554946210235, "clip_ratio/low_mean": 0.008793290238827467, "clip_ratio/low_min": 0.008793290238827467, "clip_ratio/high_mean": 0.0028409091755747795, "clip_ratio/high_max": 0.0028409091755747795, "clip_ratio/region_mean": 0.011634199414402246, "reward_total_mean": 0.9838736057281494, "reward_meter_mean": 0.9838736057281494, "reward_meter_std": 0.0036353026516735554, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9838736057281494, "reward_total_composite_std": 0.0036353026516735554, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 823.0} {"timestamp_utc": "2026-04-11T21:10:09Z", "mode": "train", "global_step": 824, "epoch": 0.031819586036453504, "loss": 0.0429, "grad_norm": 1.9315519332885742, "learning_rate": 7.5060606060606065e-06, "num_tokens": 1784630.0, "completions/mean_length": 166.375, "completions/min_length": 145.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.375, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.9164490699768066, "rewards/meter/std": 0.11904587596654892, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8527706861495972, "rewards/total_composite/std": 0.16655048727989197, "reward": 0.8527706861495972, "reward_std": 0.16655047237873077, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014335539191961288, "sampling/sampling_logp_difference/max": 3.844541072845459, "sampling/importance_sampling_ratio/min": 0.021396219730377197, "sampling/importance_sampling_ratio/mean": 0.9996809959411621, "sampling/importance_sampling_ratio/max": 1.8252487182617188, "entropy": 0.0363018496427685, "clip_ratio/low_mean": 0.007027376443147659, "clip_ratio/low_min": 0.007027376443147659, "clip_ratio/high_mean": 0.00881528900936246, "clip_ratio/high_max": 0.00881528900936246, "clip_ratio/region_mean": 0.01584266545251012, "reward_total_mean": 0.8527706861495972, "reward_meter_mean": 0.9164490699768066, "reward_meter_std": 0.11904587596654892, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8527706861495972, "reward_total_composite_std": 0.16655048727989197, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 824.0} {"timestamp_utc": "2026-04-11T21:10:19Z", "mode": "train", "global_step": 825, "epoch": 0.03185820203892493, "loss": -0.1671, "grad_norm": 1.7461193799972534, "learning_rate": 7.503030303030303e-06, "num_tokens": 1787772.0, "completions/mean_length": 302.75, "completions/min_length": 222.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 233.0, "completions/min_terminated_length": 222.0, "completions/max_terminated_length": 271.0, "rewards/meter/mean": 0.8477067947387695, "rewards/meter/std": 0.2775394320487976, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.25877460837364197, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.742363452911377, "rewards/total_composite/std": 0.3543423116207123, "reward": 0.742363452911377, "reward_std": 0.3543423116207123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02795267291367054, "sampling/sampling_logp_difference/max": 13.35202407836914, "sampling/importance_sampling_ratio/min": 1.5896065406195703e-06, "sampling/importance_sampling_ratio/mean": 0.9968683123588562, "sampling/importance_sampling_ratio/max": 1.532668113708496, "entropy": 0.030033407732844353, "clip_ratio/low_mean": 0.0018450184725224972, "clip_ratio/low_min": 0.0018450184725224972, "clip_ratio/high_mean": 0.004444972844794393, "clip_ratio/high_max": 0.004444972844794393, "clip_ratio/region_mean": 0.00628999131731689, "reward_total_mean": 0.742363452911377, "reward_meter_mean": 0.8477067947387695, "reward_meter_std": 0.2775394320487976, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.25877460837364197, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.742363452911377, "reward_total_composite_std": 0.3543423116207123, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 825.0} {"timestamp_utc": "2026-04-11T21:10:23Z", "mode": "train", "global_step": 826, "epoch": 0.03189681804139635, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.500000000000001e-06, "num_tokens": 1789412.0, "completions/mean_length": 44.0, "completions/min_length": 44.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.0, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9859310984611511, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9859310984611511, "rewards/total_composite/std": 0.0, "reward": 0.9859310984611511, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0008389691938646138, "sampling/sampling_logp_difference/max": 0.04270203784108162, "sampling/importance_sampling_ratio/min": 0.9581968784332275, "sampling/importance_sampling_ratio/mean": 1.0000182390213013, "sampling/importance_sampling_ratio/max": 1.0150346755981445, "entropy": 0.005586700281128287, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9859310984611511, "reward_meter_mean": 0.9859310984611511, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9859310984611511, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 826.0} {"timestamp_utc": "2026-04-11T21:10:28Z", "mode": "train", "global_step": 827, "epoch": 0.031935434043867776, "loss": 0.0908, "grad_norm": 4.890305042266846, "learning_rate": 7.496969696969698e-06, "num_tokens": 1791030.0, "completions/mean_length": 41.25, "completions/min_length": 39.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.25, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.0029669920913875103, "rewards/meter/std": 0.0017778158653527498, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0029669920913875103, "rewards/total_composite/std": 0.0017778158653527498, "reward": 0.0029669920913875103, "reward_std": 0.0017778158653527498, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012758270837366581, "sampling/sampling_logp_difference/max": 0.8158693313598633, "sampling/importance_sampling_ratio/min": 0.4422546923160553, "sampling/importance_sampling_ratio/mean": 0.9967164397239685, "sampling/importance_sampling_ratio/max": 1.4205405712127686, "entropy": 0.043509550858289, "clip_ratio/low_mean": 0.0026041667442768812, "clip_ratio/low_min": 0.0026041667442768812, "clip_ratio/high_mean": 0.016025641234591603, "clip_ratio/high_max": 0.016025641234591603, "clip_ratio/region_mean": 0.018629807978868484, "reward_total_mean": 0.0029669920913875103, "reward_meter_mean": 0.0029669920913875103, "reward_meter_std": 0.0017778158653527498, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0029669920913875103, "reward_total_composite_std": 0.0017778158653527498, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 827.0} {"timestamp_utc": "2026-04-11T21:10:32Z", "mode": "train", "global_step": 828, "epoch": 0.0319740500463392, "loss": 0.0048, "grad_norm": 3.8690385818481445, "learning_rate": 7.493939393939395e-06, "num_tokens": 1792724.0, "completions/mean_length": 58.75, "completions/min_length": 56.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.972812294960022, "rewards/meter/std": 0.02596760354936123, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.972812294960022, "rewards/total_composite/std": 0.02596760354936123, "reward": 0.972812294960022, "reward_std": 0.025967609137296677, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04437405616044998, "sampling/sampling_logp_difference/max": 2.798241138458252, "sampling/importance_sampling_ratio/min": 0.119936004281044, "sampling/importance_sampling_ratio/mean": 0.9989261627197266, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12588711082935333, "clip_ratio/low_mean": 0.02982456237077713, "clip_ratio/low_min": 0.02982456237077713, "clip_ratio/high_mean": 0.019190080231055617, "clip_ratio/high_max": 0.019190080231055617, "clip_ratio/region_mean": 0.04901464260183275, "reward_total_mean": 0.972812294960022, "reward_meter_mean": 0.972812294960022, "reward_meter_std": 0.02596760354936123, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.972812294960022, "reward_total_composite_std": 0.02596760354936123, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 828.0} {"timestamp_utc": "2026-04-11T21:10:37Z", "mode": "train", "global_step": 829, "epoch": 0.032012666048810624, "loss": -0.0081, "grad_norm": 3.773761034011841, "learning_rate": 7.490909090909092e-06, "num_tokens": 1794229.0, "completions/mean_length": 49.125, "completions/min_length": 47.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.2953336238861084, "rewards/meter/std": 0.012875347398221493, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.2953336238861084, "rewards/total_composite/std": 0.012875347398221493, "reward": 0.2953336238861084, "reward_std": 0.012875345535576344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02989502064883709, "sampling/sampling_logp_difference/max": 1.9005825519561768, "sampling/importance_sampling_ratio/min": 0.14948152005672455, "sampling/importance_sampling_ratio/mean": 1.0004477500915527, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09608994983136654, "clip_ratio/low_mean": 0.00781914871186018, "clip_ratio/low_min": 0.00781914871186018, "clip_ratio/high_mean": 0.012409733142703772, "clip_ratio/high_max": 0.012409733142703772, "clip_ratio/region_mean": 0.02022888185456395, "reward_total_mean": 0.2953336238861084, "reward_meter_mean": 0.2953336238861084, "reward_meter_std": 0.012875347398221493, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.2953336238861084, "reward_total_composite_std": 0.012875347398221493, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 829.0} {"timestamp_utc": "2026-04-11T21:10:42Z", "mode": "train", "global_step": 830, "epoch": 0.03205128205128205, "loss": 0.0034, "grad_norm": 2.917823076248169, "learning_rate": 7.487878787878788e-06, "num_tokens": 1796157.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9981321096420288, "rewards/meter/std": 0.0006229483988136053, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9981321096420288, "rewards/total_composite/std": 0.0006229483988136053, "reward": 0.9981321096420288, "reward_std": 0.0006229397258721292, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01399965863674879, "sampling/sampling_logp_difference/max": 0.5764389038085938, "sampling/importance_sampling_ratio/min": 0.6575833559036255, "sampling/importance_sampling_ratio/mean": 1.0055092573165894, "sampling/importance_sampling_ratio/max": 1.7796894311904907, "entropy": 0.12422750471159816, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/high_mean": 0.004098360426723957, "clip_ratio/high_max": 0.004098360426723957, "clip_ratio/region_mean": 0.008196720853447914, "reward_total_mean": 0.9981321096420288, "reward_meter_mean": 0.9981321096420288, "reward_meter_std": 0.0006229483988136053, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9981321096420288, "reward_total_composite_std": 0.0006229483988136053, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 830.0} {"timestamp_utc": "2026-04-11T21:10:51Z", "mode": "train", "global_step": 831, "epoch": 0.03208989805375347, "loss": -0.1618, "grad_norm": 1.017340898513794, "learning_rate": 7.484848484848486e-06, "num_tokens": 1797881.0, "completions/mean_length": 118.5, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 62.28571701049805, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9877695441246033, "rewards/meter/std": 0.026359617710113525, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8724490404129028, "rewards/total_composite/std": 0.3525236248970032, "reward": 0.8724490404129028, "reward_std": 0.3525235950946808, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019668839871883392, "sampling/sampling_logp_difference/max": 1.930706262588501, "sampling/importance_sampling_ratio/min": 0.14504572749137878, "sampling/importance_sampling_ratio/mean": 1.0004944801330566, "sampling/importance_sampling_ratio/max": 1.8575040102005005, "entropy": 0.05983019759878516, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.01602291688323021, "clip_ratio/high_max": 0.01602291688323021, "clip_ratio/region_mean": 0.01602291688323021, "reward_total_mean": 0.8724490404129028, "reward_meter_mean": 0.9877695441246033, "reward_meter_std": 0.026359617710113525, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8724490404129028, "reward_total_composite_std": 0.3525236248970032, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 831.0} {"timestamp_utc": "2026-04-11T21:10:58Z", "mode": "train", "global_step": 832, "epoch": 0.0321285140562249, "loss": 0.0395, "grad_norm": 2.4054229259490967, "learning_rate": 7.481818181818182e-06, "num_tokens": 1800738.0, "completions/mean_length": 168.125, "completions/min_length": 154.0, "completions/max_length": 197.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.125, "completions/min_terminated_length": 154.0, "completions/max_terminated_length": 197.0, "rewards/meter/mean": 0.9122941493988037, "rewards/meter/std": 0.1966755986213684, "rewards/count_adherence/mean": 0.9821428656578064, "rewards/count_adherence/std": 0.05050762742757797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.894844651222229, "rewards/total_composite/std": 0.19628962874412537, "reward": 0.894844651222229, "reward_std": 0.19628964364528656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006011446937918663, "sampling/sampling_logp_difference/max": 0.9638941287994385, "sampling/importance_sampling_ratio/min": 0.6423860192298889, "sampling/importance_sampling_ratio/mean": 1.0033369064331055, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03239346027839929, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0015151514671742916, "clip_ratio/high_max": 0.0015151514671742916, "clip_ratio/region_mean": 0.0015151514671742916, "reward_total_mean": 0.894844651222229, "reward_meter_mean": 0.9122941493988037, "reward_meter_std": 0.1966755986213684, "reward_count_adherence_mean": 0.9821428656578064, "reward_count_adherence_std": 0.05050762742757797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.894844651222229, "reward_total_composite_std": 0.19628962874412537, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 832.0} {"timestamp_utc": "2026-04-11T21:11:08Z", "mode": "train", "global_step": 833, "epoch": 0.03216713005869632, "loss": -0.1699, "grad_norm": 0.5217118859291077, "learning_rate": 7.47878787878788e-06, "num_tokens": 1802412.0, "completions/mean_length": 124.25, "completions/min_length": 65.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 68.85714721679688, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9432027339935303, "rewards/meter/std": 0.05009361729025841, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8400217294692993, "rewards/total_composite/std": 0.3397814929485321, "reward": 0.8400217294692993, "reward_std": 0.3397814929485321, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02556472457945347, "sampling/sampling_logp_difference/max": 1.743072509765625, "sampling/importance_sampling_ratio/min": 0.17498193681240082, "sampling/importance_sampling_ratio/mean": 0.9961550235748291, "sampling/importance_sampling_ratio/max": 1.5122870206832886, "entropy": 0.07728508254513144, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.014772114111110568, "clip_ratio/high_max": 0.014772114111110568, "clip_ratio/region_mean": 0.014772114111110568, "reward_total_mean": 0.8400217294692993, "reward_meter_mean": 0.9432027339935303, "reward_meter_std": 0.05009361729025841, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8400217294692993, "reward_total_composite_std": 0.3397814929485321, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 833.0} {"timestamp_utc": "2026-04-11T21:11:12Z", "mode": "train", "global_step": 834, "epoch": 0.032205746061167745, "loss": -0.0254, "grad_norm": 8.818110466003418, "learning_rate": 7.4757575757575765e-06, "num_tokens": 1803781.0, "completions/mean_length": 25.125, "completions/min_length": 21.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.125, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.7761045098304749, "rewards/meter/std": 0.36275652050971985, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7761045098304749, "rewards/total_composite/std": 0.36275652050971985, "reward": 0.7761045098304749, "reward_std": 0.36275652050971985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06797561049461365, "sampling/sampling_logp_difference/max": 1.617978572845459, "sampling/importance_sampling_ratio/min": 0.19829913973808289, "sampling/importance_sampling_ratio/mean": 1.000751256942749, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2838082183152437, "clip_ratio/low_mean": 0.02116402145475149, "clip_ratio/low_min": 0.02116402145475149, "clip_ratio/high_mean": 0.039262821432203054, "clip_ratio/high_max": 0.039262821432203054, "clip_ratio/region_mean": 0.060426842886954546, "reward_total_mean": 0.7761045098304749, "reward_meter_mean": 0.7761045098304749, "reward_meter_std": 0.36275652050971985, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7761045098304749, "reward_total_composite_std": 0.36275652050971985, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 834.0} {"timestamp_utc": "2026-04-11T21:11:21Z", "mode": "train", "global_step": 835, "epoch": 0.03224436206363917, "loss": 0.3939, "grad_norm": 4.391183376312256, "learning_rate": 7.472727272727274e-06, "num_tokens": 1805752.0, "completions/mean_length": 90.375, "completions/min_length": 31.0, "completions/max_length": 485.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 485.0, "rewards/meter/mean": 0.9542847871780396, "rewards/meter/std": 0.05807793140411377, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9542847871780396, "rewards/total_composite/std": 0.05807793140411377, "reward": 0.9542847871780396, "reward_std": 0.058077938854694366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11851003766059875, "sampling/sampling_logp_difference/max": 1.453650951385498, "sampling/importance_sampling_ratio/min": 0.23371544480323792, "sampling/importance_sampling_ratio/mean": 1.0212384462356567, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.090888130478561, "clip_ratio/low_mean": 0.009837420657277107, "clip_ratio/low_min": 0.009837420657277107, "clip_ratio/high_mean": 0.023271888960152864, "clip_ratio/high_max": 0.023271888960152864, "clip_ratio/region_mean": 0.03310930961742997, "reward_total_mean": 0.9542847871780396, "reward_meter_mean": 0.9542847871780396, "reward_meter_std": 0.05807793140411377, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9542847871780396, "reward_total_composite_std": 0.05807793140411377, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 835.0} {"timestamp_utc": "2026-04-11T21:11:26Z", "mode": "train", "global_step": 836, "epoch": 0.03228297806611059, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.46969696969697e-06, "num_tokens": 1807560.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006163836806081235, "sampling/sampling_logp_difference/max": 0.012493029236793518, "sampling/importance_sampling_ratio/min": 0.9980220198631287, "sampling/importance_sampling_ratio/mean": 1.0005966424942017, "sampling/importance_sampling_ratio/max": 1.0125713348388672, "entropy": 0.006207629689015448, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 836.0} {"timestamp_utc": "2026-04-11T21:11:36Z", "mode": "train", "global_step": 837, "epoch": 0.03232159406858202, "loss": -0.0569, "grad_norm": 1.1902903318405151, "learning_rate": 7.4666666666666675e-06, "num_tokens": 1809936.0, "completions/mean_length": 274.0, "completions/min_length": 86.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 131.1999969482422, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 236.0, "rewards/meter/mean": 0.4752488136291504, "rewards/meter/std": 0.4163498282432556, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.24775780737400055, "rewards/arabic_clean/mean": 0.375, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.29583215713500977, "rewards/total_composite/std": 0.42450574040412903, "reward": 0.29583215713500977, "reward_std": 0.42450571060180664, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07121021300554276, "sampling/sampling_logp_difference/max": 1.4528226852416992, "sampling/importance_sampling_ratio/min": 0.23390911519527435, "sampling/importance_sampling_ratio/mean": 1.0081943273544312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8001019135117531, "clip_ratio/low_mean": 0.007528753252699971, "clip_ratio/low_min": 0.007528753252699971, "clip_ratio/high_mean": 0.01078985957428813, "clip_ratio/high_max": 0.01078985957428813, "clip_ratio/region_mean": 0.0183186128269881, "reward_total_mean": 0.29583215713500977, "reward_meter_mean": 0.4752488136291504, "reward_meter_std": 0.4163498282432556, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.24775780737400055, "reward_arabic_clean_mean": 0.375, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.29583215713500977, "reward_total_composite_std": 0.42450574040412903, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 837.0} {"timestamp_utc": "2026-04-11T21:11:46Z", "mode": "train", "global_step": 838, "epoch": 0.03236021007105344, "loss": -0.1186, "grad_norm": 3.2613723278045654, "learning_rate": 7.463636363636364e-06, "num_tokens": 1811964.0, "completions/mean_length": 144.5, "completions/min_length": 82.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 92.00000762939453, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.5369187593460083, "rewards/meter/std": 0.37629562616348267, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5305782556533813, "rewards/total_composite/std": 0.3859613835811615, "reward": 0.5305782556533813, "reward_std": 0.3859613835811615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.053156718611717224, "sampling/sampling_logp_difference/max": 1.824774980545044, "sampling/importance_sampling_ratio/min": 0.1612539291381836, "sampling/importance_sampling_ratio/mean": 1.0106358528137207, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22520209103822708, "clip_ratio/low_mean": 0.010129818692803383, "clip_ratio/low_min": 0.010129818692803383, "clip_ratio/high_mean": 0.0170740676112473, "clip_ratio/high_max": 0.0170740676112473, "clip_ratio/region_mean": 0.027203886304050684, "reward_total_mean": 0.5305782556533813, "reward_meter_mean": 0.5369187593460083, "reward_meter_std": 0.37629562616348267, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5305782556533813, "reward_total_composite_std": 0.3859613835811615, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 838.0} {"timestamp_utc": "2026-04-11T21:11:51Z", "mode": "train", "global_step": 839, "epoch": 0.032398826073524865, "loss": -0.0243, "grad_norm": 1.519675612449646, "learning_rate": 7.460606060606061e-06, "num_tokens": 1814101.0, "completions/mean_length": 87.125, "completions/min_length": 81.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.125, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9800252914428711, "rewards/meter/std": 0.0235956609249115, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9800252914428711, "rewards/total_composite/std": 0.0235956609249115, "reward": 0.9800252914428711, "reward_std": 0.0235956609249115, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0017693947302177548, "sampling/sampling_logp_difference/max": 0.3191537857055664, "sampling/importance_sampling_ratio/min": 0.9509286284446716, "sampling/importance_sampling_ratio/mean": 1.0013376474380493, "sampling/importance_sampling_ratio/max": 1.3759629726409912, "entropy": 0.016418762621469796, "clip_ratio/low_mean": 0.0015432098880410194, "clip_ratio/low_min": 0.0015432098880410194, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0015432098880410194, "reward_total_mean": 0.9800252914428711, "reward_meter_mean": 0.9800252914428711, "reward_meter_std": 0.0235956609249115, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9800252914428711, "reward_total_composite_std": 0.0235956609249115, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 839.0} {"timestamp_utc": "2026-04-11T21:12:01Z", "mode": "train", "global_step": 840, "epoch": 0.03243744207599629, "loss": -0.3207, "grad_norm": 0.9235149025917053, "learning_rate": 7.4575757575757575e-06, "num_tokens": 1816651.0, "completions/mean_length": 478.75, "completions/min_length": 340.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.75, "completions/mean_terminated_length": 379.0, "completions/min_terminated_length": 340.0, "completions/max_terminated_length": 418.0, "rewards/meter/mean": 0.4573158323764801, "rewards/meter/std": 0.45601344108581543, "rewards/count_adherence/mean": 0.42500001192092896, "rewards/count_adherence/std": 0.2893616557121277, "rewards/arabic_clean/mean": 0.25, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.20263326168060303, "rewards/total_composite/std": 0.3752988278865814, "reward": 0.20263326168060303, "reward_std": 0.3752988278865814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009610231034457684, "sampling/sampling_logp_difference/max": 0.7336547374725342, "sampling/importance_sampling_ratio/min": 0.48015096783638, "sampling/importance_sampling_ratio/mean": 1.0045840740203857, "sampling/importance_sampling_ratio/max": 1.3668442964553833, "entropy": 0.017722988035529852, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.002230509475339204, "clip_ratio/high_max": 0.002230509475339204, "clip_ratio/region_mean": 0.002230509475339204, "reward_total_mean": 0.20263326168060303, "reward_meter_mean": 0.4573158323764801, "reward_meter_std": 0.45601344108581543, "reward_count_adherence_mean": 0.42500001192092896, "reward_count_adherence_std": 0.2893616557121277, "reward_arabic_clean_mean": 0.25, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.20263326168060303, "reward_total_composite_std": 0.3752988278865814, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 840.0} {"timestamp_utc": "2026-04-11T21:12:11Z", "mode": "train", "global_step": 841, "epoch": 0.032476058078467714, "loss": -0.1975, "grad_norm": 1.5291121006011963, "learning_rate": 7.454545454545456e-06, "num_tokens": 1818998.0, "completions/mean_length": 171.375, "completions/min_length": 114.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 122.71429443359375, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.6735749244689941, "rewards/meter/std": 0.1982181817293167, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6384245157241821, "rewards/total_composite/std": 0.2840765416622162, "reward": 0.6384245157241821, "reward_std": 0.2840765416622162, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042852405458688736, "sampling/sampling_logp_difference/max": 1.1612436771392822, "sampling/importance_sampling_ratio/min": 0.3130965530872345, "sampling/importance_sampling_ratio/mean": 1.008310317993164, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3063155012205243, "clip_ratio/low_mean": 0.007125603733584285, "clip_ratio/low_min": 0.007125603733584285, "clip_ratio/high_mean": 0.023393891751766205, "clip_ratio/high_max": 0.023393891751766205, "clip_ratio/region_mean": 0.03051949548535049, "reward_total_mean": 0.6384245157241821, "reward_meter_mean": 0.6735749244689941, "reward_meter_std": 0.1982181817293167, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6384245157241821, "reward_total_composite_std": 0.2840765416622162, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 841.0} {"timestamp_utc": "2026-04-11T21:12:18Z", "mode": "train", "global_step": 842, "epoch": 0.032514674080939145, "loss": -0.0509, "grad_norm": 0.6401534080505371, "learning_rate": 7.451515151515152e-06, "num_tokens": 1823004.0, "completions/mean_length": 277.75, "completions/min_length": 235.0, "completions/max_length": 286.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 277.75, "completions/min_terminated_length": 235.0, "completions/max_terminated_length": 286.0, "rewards/meter/mean": 0.997998833656311, "rewards/meter/std": 0.001558481715619564, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8981989622116089, "rewards/total_composite/std": 0.0014026371063664556, "reward": 0.8981989622116089, "reward_std": 0.0014026258140802383, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.001804076717235148, "sampling/sampling_logp_difference/max": 1.8155012130737305, "sampling/importance_sampling_ratio/min": 0.4627200961112976, "sampling/importance_sampling_ratio/mean": 1.000341534614563, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.004274196311598644, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8981989622116089, "reward_meter_mean": 0.997998833656311, "reward_meter_std": 0.001558481715619564, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8981989622116089, "reward_total_composite_std": 0.0014026371063664556, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 842.0} {"timestamp_utc": "2026-04-11T21:12:24Z", "mode": "train", "global_step": 843, "epoch": 0.03255329008341057, "loss": -0.0404, "grad_norm": 0.3785923719406128, "learning_rate": 7.448484848484849e-06, "num_tokens": 1825341.0, "completions/mean_length": 119.125, "completions/min_length": 106.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.125, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9673861265182495, "rewards/total_composite/std": 0.08826391398906708, "reward": 0.9673861265182495, "reward_std": 0.08826389163732529, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002283617155626416, "sampling/sampling_logp_difference/max": 0.6002688407897949, "sampling/importance_sampling_ratio/min": 0.5486640930175781, "sampling/importance_sampling_ratio/mean": 0.9997179508209229, "sampling/importance_sampling_ratio/max": 1.1165944337844849, "entropy": 0.0068072130670771, "clip_ratio/low_mean": 0.001179245300590992, "clip_ratio/low_min": 0.001179245300590992, "clip_ratio/high_mean": 0.001033057807944715, "clip_ratio/high_max": 0.001033057807944715, "clip_ratio/region_mean": 0.002212303108535707, "reward_total_mean": 0.9673861265182495, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9673861265182495, "reward_total_composite_std": 0.08826391398906708, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 843.0} {"timestamp_utc": "2026-04-11T21:12:29Z", "mode": "train", "global_step": 844, "epoch": 0.03259190608588199, "loss": 0.0515, "grad_norm": 4.047276973724365, "learning_rate": 7.445454545454546e-06, "num_tokens": 1827477.0, "completions/mean_length": 96.0, "completions/min_length": 82.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.0, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.012957379221916199, "rewards/meter/std": 0.03628787398338318, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.012957379221916199, "rewards/total_composite/std": 0.03628787398338318, "reward": 0.012957379221916199, "reward_std": 0.03628787398338318, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008930135518312454, "sampling/sampling_logp_difference/max": 1.6372274160385132, "sampling/importance_sampling_ratio/min": 0.19451862573623657, "sampling/importance_sampling_ratio/mean": 0.999650239944458, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.026015828596428037, "clip_ratio/low_mean": 0.0038265305338427424, "clip_ratio/low_min": 0.0038265305338427424, "clip_ratio/high_mean": 0.0015243901871144772, "clip_ratio/high_max": 0.0015243901871144772, "clip_ratio/region_mean": 0.00535092072095722, "reward_total_mean": 0.012957379221916199, "reward_meter_mean": 0.012957379221916199, "reward_meter_std": 0.03628787398338318, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.012957379221916199, "reward_total_composite_std": 0.03628787398338318, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 844.0} {"timestamp_utc": "2026-04-11T21:12:33Z", "mode": "train", "global_step": 845, "epoch": 0.03263052208835342, "loss": -0.0142, "grad_norm": 4.978487968444824, "learning_rate": 7.442424242424243e-06, "num_tokens": 1829046.0, "completions/mean_length": 36.125, "completions/min_length": 36.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9570483565330505, "rewards/meter/std": 0.011595988646149635, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9570483565330505, "rewards/total_composite/std": 0.011595988646149635, "reward": 0.9570483565330505, "reward_std": 0.011595983058214188, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010491784662008286, "sampling/sampling_logp_difference/max": 0.6294360160827637, "sampling/importance_sampling_ratio/min": 0.5328922867774963, "sampling/importance_sampling_ratio/mean": 1.00355064868927, "sampling/importance_sampling_ratio/max": 1.4467881917953491, "entropy": 0.0663958489894867, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.010135134682059288, "clip_ratio/high_max": 0.010135134682059288, "clip_ratio/region_mean": 0.010135134682059288, "reward_total_mean": 0.9570483565330505, "reward_meter_mean": 0.9570483565330505, "reward_meter_std": 0.011595988646149635, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9570483565330505, "reward_total_composite_std": 0.011595988646149635, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 845.0} {"timestamp_utc": "2026-04-11T21:12:38Z", "mode": "train", "global_step": 846, "epoch": 0.03266913809082484, "loss": 0.0554, "grad_norm": 7.340315341949463, "learning_rate": 7.439393939393939e-06, "num_tokens": 1830756.0, "completions/mean_length": 47.75, "completions/min_length": 43.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.013893251307308674, "rewards/meter/std": 0.015486767515540123, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.013893251307308674, "rewards/total_composite/std": 0.015486767515540123, "reward": 0.013893251307308674, "reward_std": 0.015486767515540123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.044018059968948364, "sampling/sampling_logp_difference/max": 1.2990374565124512, "sampling/importance_sampling_ratio/min": 0.272794246673584, "sampling/importance_sampling_ratio/mean": 1.0058790445327759, "sampling/importance_sampling_ratio/max": 1.8971904516220093, "entropy": 0.16840760409832, "clip_ratio/low_mean": 0.01568623329512775, "clip_ratio/low_min": 0.01568623329512775, "clip_ratio/high_mean": 0.022192253498360515, "clip_ratio/high_max": 0.022192253498360515, "clip_ratio/region_mean": 0.037878486793488264, "reward_total_mean": 0.013893251307308674, "reward_meter_mean": 0.013893251307308674, "reward_meter_std": 0.015486767515540123, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.013893251307308674, "reward_total_composite_std": 0.015486767515540123, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 846.0} {"timestamp_utc": "2026-04-11T21:12:42Z", "mode": "train", "global_step": 847, "epoch": 0.032707754093296265, "loss": -0.0123, "grad_norm": 11.845059394836426, "learning_rate": 7.4363636363636375e-06, "num_tokens": 1832303.0, "completions/mean_length": 34.375, "completions/min_length": 32.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8031737804412842, "rewards/meter/std": 0.21358220279216766, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8031737804412842, "rewards/total_composite/std": 0.21358220279216766, "reward": 0.8031737804412842, "reward_std": 0.21358218789100647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05371801182627678, "sampling/sampling_logp_difference/max": 0.9102482795715332, "sampling/importance_sampling_ratio/min": 0.40242430567741394, "sampling/importance_sampling_ratio/mean": 1.004541039466858, "sampling/importance_sampling_ratio/max": 1.5834014415740967, "entropy": 0.24202715791761875, "clip_ratio/low_mean": 0.007352941203862429, "clip_ratio/low_min": 0.007352941203862429, "clip_ratio/high_mean": 0.021566784707829356, "clip_ratio/high_max": 0.021566784707829356, "clip_ratio/region_mean": 0.028919725911691785, "reward_total_mean": 0.8031737804412842, "reward_meter_mean": 0.8031737804412842, "reward_meter_std": 0.21358220279216766, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8031737804412842, "reward_total_composite_std": 0.21358220279216766, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 847.0} {"timestamp_utc": "2026-04-11T21:12:47Z", "mode": "train", "global_step": 848, "epoch": 0.03274637009576769, "loss": 0.0682, "grad_norm": 9.50920295715332, "learning_rate": 7.433333333333334e-06, "num_tokens": 1833837.0, "completions/mean_length": 32.75, "completions/min_length": 27.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.75, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.7122796177864075, "rewards/meter/std": 0.41455161571502686, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7122796177864075, "rewards/total_composite/std": 0.41455161571502686, "reward": 0.7122796177864075, "reward_std": 0.41455161571502686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05715809762477875, "sampling/sampling_logp_difference/max": 0.8534235954284668, "sampling/importance_sampling_ratio/min": 0.4633271396160126, "sampling/importance_sampling_ratio/mean": 1.0182219743728638, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3148685749620199, "clip_ratio/low_mean": 0.016945773735642433, "clip_ratio/low_min": 0.016945773735642433, "clip_ratio/high_mean": 0.018978851148858666, "clip_ratio/high_max": 0.018978851148858666, "clip_ratio/region_mean": 0.0359246248845011, "reward_total_mean": 0.7122796177864075, "reward_meter_mean": 0.7122796177864075, "reward_meter_std": 0.41455161571502686, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7122796177864075, "reward_total_composite_std": 0.41455161571502686, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 848.0} {"timestamp_utc": "2026-04-11T21:12:52Z", "mode": "train", "global_step": 849, "epoch": 0.03278498609823911, "loss": -0.0012, "grad_norm": 4.006454944610596, "learning_rate": 7.430303030303031e-06, "num_tokens": 1835618.0, "completions/mean_length": 69.625, "completions/min_length": 67.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9579675793647766, "rewards/meter/std": 0.014622213318943977, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9579675793647766, "rewards/total_composite/std": 0.014622213318943977, "reward": 0.9579675793647766, "reward_std": 0.014622238464653492, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021169880405068398, "sampling/sampling_logp_difference/max": 1.017970085144043, "sampling/importance_sampling_ratio/min": 0.3613276779651642, "sampling/importance_sampling_ratio/mean": 0.9958710074424744, "sampling/importance_sampling_ratio/max": 1.4254369735717773, "entropy": 0.08608170412480831, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/high_mean": 0.008808409329503775, "clip_ratio/high_max": 0.008808409329503775, "clip_ratio/region_mean": 0.010594123625196517, "reward_total_mean": 0.9579675793647766, "reward_meter_mean": 0.9579675793647766, "reward_meter_std": 0.014622213318943977, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9579675793647766, "reward_total_composite_std": 0.014622213318943977, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 849.0} {"timestamp_utc": "2026-04-11T21:12:57Z", "mode": "train", "global_step": 850, "epoch": 0.03282360210071054, "loss": 0.0388, "grad_norm": 11.478070259094238, "learning_rate": 7.4272727272727275e-06, "num_tokens": 1837462.0, "completions/mean_length": 72.5, "completions/min_length": 70.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.5, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.8595837950706482, "rewards/meter/std": 0.2809332311153412, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8595837950706482, "rewards/total_composite/std": 0.2809332311153412, "reward": 0.8595837950706482, "reward_std": 0.2809332311153412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03292540833353996, "sampling/sampling_logp_difference/max": 2.2321743965148926, "sampling/importance_sampling_ratio/min": 0.10729487985372543, "sampling/importance_sampling_ratio/mean": 1.0040628910064697, "sampling/importance_sampling_ratio/max": 1.9754526615142822, "entropy": 0.1720542022958398, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/high_mean": 0.02274737018160522, "clip_ratio/high_max": 0.02274737018160522, "clip_ratio/region_mean": 0.027555062668398023, "reward_total_mean": 0.8595837950706482, "reward_meter_mean": 0.8595837950706482, "reward_meter_std": 0.2809332311153412, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8595837950706482, "reward_total_composite_std": 0.2809332311153412, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 850.0} {"timestamp_utc": "2026-04-11T21:14:24Z", "mode": "eval", "global_step": 850, "epoch": 0.03282360210071054, "eval_loss": NaN, "eval_runtime": 87.1966, "eval_samples_per_second": 1.193, "eval_steps_per_second": 0.149, "eval_num_tokens": 1837462.0, "eval_completions/mean_length": 238.41346153846155, "eval_completions/min_length": 63.23076923076923, "eval_completions/max_length": 468.38461538461536, "eval_completions/clipped_ratio": 0.125, "eval_completions/mean_terminated_length": 200.1790325458233, "eval_completions/min_terminated_length": 63.23076923076923, "eval_completions/max_terminated_length": 397.0, "eval_rewards/meter/mean": 0.5329499175915351, "eval_rewards/meter/std": 0.40038574085785794, "eval_rewards/count_adherence/mean": 0.9358828663825989, "eval_rewards/count_adherence/std": 0.1135323582073817, "eval_rewards/arabic_clean/mean": 0.8942307692307693, "eval_rewards/arabic_clean/std": 0.2240231060064756, "eval_rewards/total_composite/mean": 0.4906711486669687, "eval_rewards/total_composite/std": 0.39612421508018786, "eval_reward": 0.4906711486669687, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.008215394802391529, "eval_sampling/sampling_logp_difference/max": 0.7172192701926599, "eval_sampling/importance_sampling_ratio/min": 0.502099701991448, "eval_sampling/importance_sampling_ratio/mean": 1.0027639590776884, "eval_sampling/importance_sampling_ratio/max": 1.319464399264409, "eval_entropy": 0.12312392661204705, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4906711486669687, "eval_reward_meter_mean": 0.5329499175915351, "eval_reward_meter_std": 0.40038574085785794, "eval_reward_count_adherence_mean": 0.9358828663825989, "eval_reward_count_adherence_std": 0.1135323582073817, "eval_reward_arabic_clean_mean": 0.8942307692307693, "eval_reward_arabic_clean_std": 0.2240231060064756, "eval_reward_total_composite_mean": 0.4906711486669687, "eval_reward_total_composite_std": 0.39612421508018786, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 850.0} {"timestamp_utc": "2026-04-11T21:14:33Z", "mode": "train", "global_step": 851, "epoch": 0.03286221810318196, "loss": -0.0027, "grad_norm": 1.7194498777389526, "learning_rate": 7.424242424242425e-06, "num_tokens": 1839518.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9981642961502075, "rewards/meter/std": 0.0012099517043679953, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9981642961502075, "rewards/total_composite/std": 0.0012099517043679953, "reward": 0.9981642961502075, "reward_std": 0.0012099517043679953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0026090925093740225, "sampling/sampling_logp_difference/max": 1.235957145690918, "sampling/importance_sampling_ratio/min": 0.29055652022361755, "sampling/importance_sampling_ratio/mean": 0.9996359944343567, "sampling/importance_sampling_ratio/max": 1.049167275428772, "entropy": 0.00842546095373109, "clip_ratio/low_mean": 0.0013736264081671834, "clip_ratio/low_min": 0.0013736264081671834, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0013736264081671834, "reward_total_mean": 0.9981642961502075, "reward_meter_mean": 0.9981642961502075, "reward_meter_std": 0.0012099517043679953, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9981642961502075, "reward_total_composite_std": 0.0012099517043679953, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 851.0} {"timestamp_utc": "2026-04-11T21:14:38Z", "mode": "train", "global_step": 852, "epoch": 0.032900834105653386, "loss": 0.0241, "grad_norm": 5.6886396408081055, "learning_rate": 7.421212121212121e-06, "num_tokens": 1841633.0, "completions/mean_length": 102.375, "completions/min_length": 98.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.375, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9957592487335205, "rewards/meter/std": 0.007790826261043549, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9957592487335205, "rewards/total_composite/std": 0.007790826261043549, "reward": 0.9957592487335205, "reward_std": 0.007790825795382261, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015552692115306854, "sampling/sampling_logp_difference/max": 2.4417948722839355, "sampling/importance_sampling_ratio/min": 0.26987552642822266, "sampling/importance_sampling_ratio/mean": 1.002767562866211, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05315098259598017, "clip_ratio/low_mean": 0.004608591319993138, "clip_ratio/low_min": 0.004608591319993138, "clip_ratio/high_mean": 0.0012376237427815795, "clip_ratio/high_max": 0.0012376237427815795, "clip_ratio/region_mean": 0.005846215062774718, "reward_total_mean": 0.9957592487335205, "reward_meter_mean": 0.9957592487335205, "reward_meter_std": 0.007790826261043549, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9957592487335205, "reward_total_composite_std": 0.007790826261043549, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 852.0} {"timestamp_utc": "2026-04-11T21:14:43Z", "mode": "train", "global_step": 853, "epoch": 0.03293945010812481, "loss": 0.0129, "grad_norm": 7.263408184051514, "learning_rate": 7.4181818181818185e-06, "num_tokens": 1843388.0, "completions/mean_length": 57.375, "completions/min_length": 56.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7365479469299316, "rewards/meter/std": 0.41198429465293884, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7365479469299316, "rewards/total_composite/std": 0.41198429465293884, "reward": 0.7365479469299316, "reward_std": 0.41198426485061646, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027241496369242668, "sampling/sampling_logp_difference/max": 1.7325024604797363, "sampling/importance_sampling_ratio/min": 0.17684131860733032, "sampling/importance_sampling_ratio/mean": 1.0001236200332642, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13850455731153488, "clip_ratio/low_mean": 0.006355931982398033, "clip_ratio/low_min": 0.006355931982398033, "clip_ratio/high_mean": 0.012690028874203563, "clip_ratio/high_max": 0.012690028874203563, "clip_ratio/region_mean": 0.019045960856601596, "reward_total_mean": 0.7365479469299316, "reward_meter_mean": 0.7365479469299316, "reward_meter_std": 0.41198429465293884, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7365479469299316, "reward_total_composite_std": 0.41198429465293884, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 853.0} {"timestamp_utc": "2026-04-11T21:14:48Z", "mode": "train", "global_step": 854, "epoch": 0.032978066110596234, "loss": 0.0155, "grad_norm": 7.064536094665527, "learning_rate": 7.415151515151515e-06, "num_tokens": 1845010.0, "completions/mean_length": 71.75, "completions/min_length": 69.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.75, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.3246036767959595, "rewards/meter/std": 0.07094007730484009, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3246036767959595, "rewards/total_composite/std": 0.07094007730484009, "reward": 0.3246036767959595, "reward_std": 0.07094008475542068, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032428670674562454, "sampling/sampling_logp_difference/max": 1.571696162223816, "sampling/importance_sampling_ratio/min": 0.20769259333610535, "sampling/importance_sampling_ratio/mean": 1.0040819644927979, "sampling/importance_sampling_ratio/max": 1.735855221748352, "entropy": 0.180419085547328, "clip_ratio/low_mean": 0.005138941807672381, "clip_ratio/low_min": 0.005138941807672381, "clip_ratio/high_mean": 0.015700483229011297, "clip_ratio/high_max": 0.015700483229011297, "clip_ratio/region_mean": 0.02083942503668368, "reward_total_mean": 0.3246036767959595, "reward_meter_mean": 0.3246036767959595, "reward_meter_std": 0.07094007730484009, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3246036767959595, "reward_total_composite_std": 0.07094007730484009, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 854.0} {"timestamp_utc": "2026-04-11T21:14:58Z", "mode": "train", "global_step": 855, "epoch": 0.03301668211306766, "loss": -0.1759, "grad_norm": 0.7583585381507874, "learning_rate": 7.412121212121213e-06, "num_tokens": 1846714.0, "completions/mean_length": 128.0, "completions/min_length": 71.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.14286041259766, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8446406126022339, "rewards/meter/std": 0.3414607644081116, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8446406126022339, "rewards/total_composite/std": 0.3414607644081116, "reward": 0.8446406126022339, "reward_std": 0.3414607644081116, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01571240834891796, "sampling/sampling_logp_difference/max": 0.8787860870361328, "sampling/importance_sampling_ratio/min": 0.41528674960136414, "sampling/importance_sampling_ratio/mean": 1.0035755634307861, "sampling/importance_sampling_ratio/max": 1.6814461946487427, "entropy": 0.07431852375157177, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0069226446794345975, "clip_ratio/high_max": 0.0069226446794345975, "clip_ratio/region_mean": 0.0069226446794345975, "reward_total_mean": 0.8446406126022339, "reward_meter_mean": 0.8446406126022339, "reward_meter_std": 0.3414607644081116, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8446406126022339, "reward_total_composite_std": 0.3414607644081116, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 855.0} {"timestamp_utc": "2026-04-11T21:15:04Z", "mode": "train", "global_step": 856, "epoch": 0.03305529811553908, "loss": 0.014, "grad_norm": 3.5409133434295654, "learning_rate": 7.40909090909091e-06, "num_tokens": 1849322.0, "completions/mean_length": 139.0, "completions/min_length": 131.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.0, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.9776046872138977, "rewards/meter/std": 0.03895730525255203, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9776046872138977, "rewards/total_composite/std": 0.03895730525255203, "reward": 0.9776046872138977, "reward_std": 0.03895730897784233, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02131645753979683, "sampling/sampling_logp_difference/max": 1.5166456699371338, "sampling/importance_sampling_ratio/min": 0.21944674849510193, "sampling/importance_sampling_ratio/mean": 1.0038254261016846, "sampling/importance_sampling_ratio/max": 1.878424882888794, "entropy": 0.10227752942591906, "clip_ratio/low_mean": 0.004401408368721604, "clip_ratio/low_min": 0.004401408368721604, "clip_ratio/high_mean": 0.00995411560870707, "clip_ratio/high_max": 0.00995411560870707, "clip_ratio/region_mean": 0.014355523977428675, "reward_total_mean": 0.9776046872138977, "reward_meter_mean": 0.9776046872138977, "reward_meter_std": 0.03895730525255203, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9776046872138977, "reward_total_composite_std": 0.03895730525255203, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 856.0} {"timestamp_utc": "2026-04-11T21:15:11Z", "mode": "train", "global_step": 857, "epoch": 0.033093914118010506, "loss": 0.0653, "grad_norm": 3.6630215644836426, "learning_rate": 7.406060606060607e-06, "num_tokens": 1851515.0, "completions/mean_length": 118.125, "completions/min_length": 108.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.125, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.440204918384552, "rewards/meter/std": 0.45058155059814453, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.440204918384552, "rewards/total_composite/std": 0.45058155059814453, "reward": 0.440204918384552, "reward_std": 0.45058152079582214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040608666837215424, "sampling/sampling_logp_difference/max": 2.0511913299560547, "sampling/importance_sampling_ratio/min": 0.12858162820339203, "sampling/importance_sampling_ratio/mean": 1.0037841796875, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2076131235808134, "clip_ratio/low_mean": 0.014009463367983699, "clip_ratio/low_min": 0.014009463367983699, "clip_ratio/high_mean": 0.011188361560925841, "clip_ratio/high_max": 0.011188361560925841, "clip_ratio/region_mean": 0.02519782492890954, "reward_total_mean": 0.440204918384552, "reward_meter_mean": 0.440204918384552, "reward_meter_std": 0.45058155059814453, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.440204918384552, "reward_total_composite_std": 0.45058155059814453, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 857.0} {"timestamp_utc": "2026-04-11T21:15:17Z", "mode": "train", "global_step": 858, "epoch": 0.03313253012048193, "loss": 0.0181, "grad_norm": 4.55625057220459, "learning_rate": 7.403030303030304e-06, "num_tokens": 1853788.0, "completions/mean_length": 115.125, "completions/min_length": 106.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.125, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.012467301450669765, "rewards/meter/std": 0.008195622824132442, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.012467301450669765, "rewards/total_composite/std": 0.008195622824132442, "reward": 0.012467301450669765, "reward_std": 0.008195622824132442, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03302113339304924, "sampling/sampling_logp_difference/max": 2.4161791801452637, "sampling/importance_sampling_ratio/min": 0.08926202356815338, "sampling/importance_sampling_ratio/mean": 1.000256061553955, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1268502762541175, "clip_ratio/low_mean": 0.013920615892857313, "clip_ratio/low_min": 0.013920615892857313, "clip_ratio/high_mean": 0.0055547208758071065, "clip_ratio/high_max": 0.0055547208758071065, "clip_ratio/region_mean": 0.01947533676866442, "reward_total_mean": 0.012467301450669765, "reward_meter_mean": 0.012467301450669765, "reward_meter_std": 0.008195622824132442, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.012467301450669765, "reward_total_composite_std": 0.008195622824132442, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 858.0} {"timestamp_utc": "2026-04-11T21:15:27Z", "mode": "train", "global_step": 859, "epoch": 0.033171146122953354, "loss": -0.0959, "grad_norm": 1.359696865081787, "learning_rate": 7.4e-06, "num_tokens": 1855915.0, "completions/mean_length": 216.875, "completions/min_length": 113.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 118.5, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.30425071716308594, "rewards/meter/std": 0.37213876843452454, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.30108019709587097, "rewards/total_composite/std": 0.3749656677246094, "reward": 0.30108019709587097, "reward_std": 0.374965637922287, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030237801373004913, "sampling/sampling_logp_difference/max": 2.4234325885772705, "sampling/importance_sampling_ratio/min": 0.08861690759658813, "sampling/importance_sampling_ratio/mean": 1.0018775463104248, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08072960190474987, "clip_ratio/low_mean": 0.010531508014537394, "clip_ratio/low_min": 0.010531508014537394, "clip_ratio/high_mean": 0.0010593220358714461, "clip_ratio/high_max": 0.0010593220358714461, "clip_ratio/region_mean": 0.01159083005040884, "reward_total_mean": 0.30108019709587097, "reward_meter_mean": 0.30425071716308594, "reward_meter_std": 0.37213876843452454, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.30108019709587097, "reward_total_composite_std": 0.3749656677246094, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 859.0} {"timestamp_utc": "2026-04-11T21:15:31Z", "mode": "train", "global_step": 860, "epoch": 0.03320976212542478, "loss": 0.0259, "grad_norm": 10.950078964233398, "learning_rate": 7.396969696969698e-06, "num_tokens": 1857306.0, "completions/mean_length": 34.875, "completions/min_length": 31.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.875, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.4986366033554077, "rewards/meter/std": 0.3628239035606384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.3853728771209717, "rewards/total_composite/std": 0.35885992646217346, "reward": 0.3853728771209717, "reward_std": 0.35885992646217346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1037890762090683, "sampling/sampling_logp_difference/max": 1.2969651222229004, "sampling/importance_sampling_ratio/min": 0.27336013317108154, "sampling/importance_sampling_ratio/mean": 0.9990217089653015, "sampling/importance_sampling_ratio/max": 1.9907257556915283, "entropy": 0.5616006217896938, "clip_ratio/low_mean": 0.03183652367442846, "clip_ratio/low_min": 0.03183652367442846, "clip_ratio/high_mean": 0.040542286820709705, "clip_ratio/high_max": 0.040542286820709705, "clip_ratio/region_mean": 0.07237881049513817, "reward_total_mean": 0.3853728771209717, "reward_meter_mean": 0.4986366033554077, "reward_meter_std": 0.3628239035606384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.3853728771209717, "reward_total_composite_std": 0.35885992646217346, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 860.0} {"timestamp_utc": "2026-04-11T21:15:36Z", "mode": "train", "global_step": 861, "epoch": 0.0332483781278962, "loss": -0.016, "grad_norm": 8.174256324768066, "learning_rate": 7.393939393939395e-06, "num_tokens": 1858798.0, "completions/mean_length": 39.5, "completions/min_length": 36.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9748528003692627, "rewards/meter/std": 0.051897767931222916, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9748528003692627, "rewards/total_composite/std": 0.051897767931222916, "reward": 0.9748528003692627, "reward_std": 0.05189777910709381, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035488735884428024, "sampling/sampling_logp_difference/max": 1.1401195526123047, "sampling/importance_sampling_ratio/min": 0.31978079676628113, "sampling/importance_sampling_ratio/mean": 0.9993390440940857, "sampling/importance_sampling_ratio/max": 1.4244297742843628, "entropy": 0.14569186326116323, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.031030334066599607, "clip_ratio/high_max": 0.031030334066599607, "clip_ratio/region_mean": 0.031030334066599607, "reward_total_mean": 0.9748528003692627, "reward_meter_mean": 0.9748528003692627, "reward_meter_std": 0.051897767931222916, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9748528003692627, "reward_total_composite_std": 0.051897767931222916, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 861.0} {"timestamp_utc": "2026-04-11T21:15:41Z", "mode": "train", "global_step": 862, "epoch": 0.03328699413036763, "loss": 0.0553, "grad_norm": 4.272212505340576, "learning_rate": 7.390909090909092e-06, "num_tokens": 1860597.0, "completions/mean_length": 61.875, "completions/min_length": 56.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.26651522517204285, "rewards/meter/std": 0.42873692512512207, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.26651522517204285, "rewards/total_composite/std": 0.42873692512512207, "reward": 0.26651522517204285, "reward_std": 0.4287368953227997, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025758842006325722, "sampling/sampling_logp_difference/max": 1.1047568321228027, "sampling/importance_sampling_ratio/min": 0.33129143714904785, "sampling/importance_sampling_ratio/mean": 1.0066983699798584, "sampling/importance_sampling_ratio/max": 1.6077079772949219, "entropy": 0.1662331586703658, "clip_ratio/low_mean": 0.011781753972172737, "clip_ratio/low_min": 0.011781753972172737, "clip_ratio/high_mean": 0.004425125429406762, "clip_ratio/high_max": 0.004425125429406762, "clip_ratio/region_mean": 0.0162068794015795, "reward_total_mean": 0.26651522517204285, "reward_meter_mean": 0.26651522517204285, "reward_meter_std": 0.42873692512512207, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.26651522517204285, "reward_total_composite_std": 0.42873692512512207, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 862.0} {"timestamp_utc": "2026-04-11T21:15:45Z", "mode": "train", "global_step": 863, "epoch": 0.03332561013283905, "loss": 0.0863, "grad_norm": 19.171913146972656, "learning_rate": 7.3878787878787885e-06, "num_tokens": 1862168.0, "completions/mean_length": 30.375, "completions/min_length": 27.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.375, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.4400164783000946, "rewards/meter/std": 0.45030924677848816, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4400164783000946, "rewards/total_composite/std": 0.45030924677848816, "reward": 0.4400164783000946, "reward_std": 0.45030924677848816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05614474043250084, "sampling/sampling_logp_difference/max": 1.2786493301391602, "sampling/importance_sampling_ratio/min": 0.2784130871295929, "sampling/importance_sampling_ratio/mean": 1.0034319162368774, "sampling/importance_sampling_ratio/max": 1.7619303464889526, "entropy": 0.21260146889835596, "clip_ratio/low_mean": 0.037997160106897354, "clip_ratio/low_min": 0.037997160106897354, "clip_ratio/high_mean": 0.004629629664123058, "clip_ratio/high_max": 0.004629629664123058, "clip_ratio/region_mean": 0.04262678977102041, "reward_total_mean": 0.4400164783000946, "reward_meter_mean": 0.4400164783000946, "reward_meter_std": 0.45030924677848816, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4400164783000946, "reward_total_composite_std": 0.45030924677848816, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 863.0} {"timestamp_utc": "2026-04-11T21:15:55Z", "mode": "train", "global_step": 864, "epoch": 0.033364226135310475, "loss": -0.045, "grad_norm": 2.6863462924957275, "learning_rate": 7.384848484848486e-06, "num_tokens": 1863964.0, "completions/mean_length": 184.5, "completions/min_length": 68.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 75.33333587646484, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.3107237219810486, "rewards/meter/std": 0.42843642830848694, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.30930203199386597, "rewards/total_composite/std": 0.4295937418937683, "reward": 0.30930203199386597, "reward_std": 0.4295937418937683, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05242498591542244, "sampling/sampling_logp_difference/max": 0.9521334171295166, "sampling/importance_sampling_ratio/min": 0.42793694138526917, "sampling/importance_sampling_ratio/mean": 1.0233484506607056, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26837953738868237, "clip_ratio/low_mean": 0.030241255182772875, "clip_ratio/low_min": 0.030241255182772875, "clip_ratio/high_mean": 0.0073529413202777505, "clip_ratio/high_max": 0.0073529413202777505, "clip_ratio/region_mean": 0.037594196503050625, "reward_total_mean": 0.30930203199386597, "reward_meter_mean": 0.3107237219810486, "reward_meter_std": 0.42843642830848694, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.30930203199386597, "reward_total_composite_std": 0.4295937418937683, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 864.0} {"timestamp_utc": "2026-04-11T21:16:04Z", "mode": "train", "global_step": 865, "epoch": 0.0334028421377819, "loss": -0.2645, "grad_norm": 0.7457296252250671, "learning_rate": 7.381818181818182e-06, "num_tokens": 1867081.0, "completions/mean_length": 333.625, "completions/min_length": 259.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 274.16668701171875, "completions/min_terminated_length": 259.0, "completions/max_terminated_length": 304.0, "rewards/meter/mean": 0.5156766176223755, "rewards/meter/std": 0.35733509063720703, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.405046284198761, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5026091933250427, "rewards/total_composite/std": 0.376709908246994, "reward": 0.5026091933250427, "reward_std": 0.376709908246994, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008062604814767838, "sampling/sampling_logp_difference/max": 0.9383625984191895, "sampling/importance_sampling_ratio/min": 0.48332223296165466, "sampling/importance_sampling_ratio/mean": 1.0038177967071533, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03134010429494083, "clip_ratio/low_mean": 0.0013483551447279751, "clip_ratio/low_min": 0.0013483551447279751, "clip_ratio/high_mean": 0.002828874043188989, "clip_ratio/high_max": 0.002828874043188989, "clip_ratio/region_mean": 0.004177229187916964, "reward_total_mean": 0.5026091933250427, "reward_meter_mean": 0.5156766176223755, "reward_meter_std": 0.35733509063720703, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.405046284198761, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5026091933250427, "reward_total_composite_std": 0.376709908246994, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 865.0} {"timestamp_utc": "2026-04-11T21:16:09Z", "mode": "train", "global_step": 866, "epoch": 0.03344145814025332, "loss": 0.0068, "grad_norm": 5.245387554168701, "learning_rate": 7.378787878787879e-06, "num_tokens": 1868780.0, "completions/mean_length": 70.375, "completions/min_length": 67.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9889371991157532, "rewards/meter/std": 0.014782732352614403, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9292163848876953, "rewards/total_composite/std": 0.18251283466815948, "reward": 0.9292163848876953, "reward_std": 0.1825128197669983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026045849546790123, "sampling/sampling_logp_difference/max": 1.7327086925506592, "sampling/importance_sampling_ratio/min": 0.17680484056472778, "sampling/importance_sampling_ratio/mean": 1.0011097192764282, "sampling/importance_sampling_ratio/max": 1.6229914426803589, "entropy": 0.1178249791264534, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/high_mean": 0.012193023809231818, "clip_ratio/high_max": 0.012193023809231818, "clip_ratio/region_mean": 0.01397873810492456, "reward_total_mean": 0.9292163848876953, "reward_meter_mean": 0.9889371991157532, "reward_meter_std": 0.014782732352614403, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9292163848876953, "reward_total_composite_std": 0.18251283466815948, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 866.0} {"timestamp_utc": "2026-04-11T21:16:14Z", "mode": "train", "global_step": 867, "epoch": 0.03348007414272475, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.375757575757576e-06, "num_tokens": 1870332.0, "completions/mean_length": 31.0, "completions/min_length": 31.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9982872605323792, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9982872605323792, "rewards/total_composite/std": 0.0, "reward": 0.9982872605323792, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0027503161691129208, "sampling/sampling_logp_difference/max": 0.1928137242794037, "sampling/importance_sampling_ratio/min": 0.8246355652809143, "sampling/importance_sampling_ratio/mean": 0.9998703002929688, "sampling/importance_sampling_ratio/max": 1.0706051588058472, "entropy": 0.023529303492978215, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9982872605323792, "reward_meter_mean": 0.9982872605323792, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9982872605323792, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 867.0} {"timestamp_utc": "2026-04-11T21:16:21Z", "mode": "train", "global_step": 868, "epoch": 0.03351869014519617, "loss": -0.0, "grad_norm": 0.11152661591768265, "learning_rate": 7.372727272727274e-06, "num_tokens": 1873572.0, "completions/mean_length": 226.0, "completions/min_length": 226.0, "completions/max_length": 226.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 226.0, "completions/min_terminated_length": 226.0, "completions/max_terminated_length": 226.0, "rewards/meter/mean": 0.9983253479003906, "rewards/meter/std": 0.00010776949056889862, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9983253479003906, "rewards/total_composite/std": 0.00010776949056889862, "reward": 0.9983253479003906, "reward_std": 0.00010777551506180316, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0009319577366113663, "sampling/sampling_logp_difference/max": 0.596361517906189, "sampling/importance_sampling_ratio/min": 0.5508120656013489, "sampling/importance_sampling_ratio/mean": 0.9998047351837158, "sampling/importance_sampling_ratio/max": 1.0677015781402588, "entropy": 0.003107917174929753, "clip_ratio/low_mean": 0.0005530973430722952, "clip_ratio/low_min": 0.0005530973430722952, "clip_ratio/high_mean": 0.0005530973430722952, "clip_ratio/high_max": 0.0005530973430722952, "clip_ratio/region_mean": 0.0011061946861445904, "reward_total_mean": 0.9983253479003906, "reward_meter_mean": 0.9983253479003906, "reward_meter_std": 0.00010776949056889862, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9983253479003906, "reward_total_composite_std": 0.00010776949056889862, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 868.0} {"timestamp_utc": "2026-04-11T21:16:31Z", "mode": "train", "global_step": 869, "epoch": 0.033557306147667595, "loss": -0.2731, "grad_norm": 0.36766162514686584, "learning_rate": 7.36969696969697e-06, "num_tokens": 1876874.0, "completions/mean_length": 289.75, "completions/min_length": 233.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 258.0, "completions/min_terminated_length": 233.0, "completions/max_terminated_length": 275.0, "rewards/meter/mean": 0.836525022983551, "rewards/meter/std": 0.3188902735710144, "rewards/count_adherence/mean": 0.890625, "rewards/count_adherence/std": 0.26252126693725586, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8159967660903931, "rewards/total_composite/std": 0.3426204025745392, "reward": 0.8159967660903931, "reward_std": 0.3426204025745392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00748072937130928, "sampling/sampling_logp_difference/max": 1.9644577503204346, "sampling/importance_sampling_ratio/min": 0.1402319073677063, "sampling/importance_sampling_ratio/mean": 0.9986810088157654, "sampling/importance_sampling_ratio/max": 1.2007663249969482, "entropy": 0.02872346807271242, "clip_ratio/low_mean": 0.0005364806856960058, "clip_ratio/low_min": 0.0005364806856960058, "clip_ratio/high_mean": 0.004278188047464937, "clip_ratio/high_max": 0.004278188047464937, "clip_ratio/region_mean": 0.004814668733160943, "reward_total_mean": 0.8159967660903931, "reward_meter_mean": 0.836525022983551, "reward_meter_std": 0.3188902735710144, "reward_count_adherence_mean": 0.890625, "reward_count_adherence_std": 0.26252126693725586, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8159967660903931, "reward_total_composite_std": 0.3426204025745392, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 869.0} {"timestamp_utc": "2026-04-11T21:16:36Z", "mode": "train", "global_step": 870, "epoch": 0.03359592215013902, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.3666666666666676e-06, "num_tokens": 1879314.0, "completions/mean_length": 121.0, "completions/min_length": 121.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.0, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9982872605323792, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9982872605323792, "rewards/total_composite/std": 0.0, "reward": 0.9982872605323792, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0004436108865775168, "sampling/sampling_logp_difference/max": 0.0253915935754776, "sampling/importance_sampling_ratio/min": 0.9974165558815002, "sampling/importance_sampling_ratio/mean": 1.000435471534729, "sampling/importance_sampling_ratio/max": 1.025716781616211, "entropy": 0.004498630034504458, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9982872605323792, "reward_meter_mean": 0.9982872605323792, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9982872605323792, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 870.0} {"timestamp_utc": "2026-04-11T21:16:41Z", "mode": "train", "global_step": 871, "epoch": 0.033634538152610444, "loss": 0.0591, "grad_norm": 3.21658992767334, "learning_rate": 7.363636363636364e-06, "num_tokens": 1881171.0, "completions/mean_length": 63.125, "completions/min_length": 61.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9981216788291931, "rewards/meter/std": 0.000908432062715292, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9981216788291931, "rewards/total_composite/std": 0.000908432062715292, "reward": 0.9981216788291931, "reward_std": 0.0009084459743462503, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010571952909231186, "sampling/sampling_logp_difference/max": 1.0176496505737305, "sampling/importance_sampling_ratio/min": 0.3614434599876404, "sampling/importance_sampling_ratio/mean": 0.9997851252555847, "sampling/importance_sampling_ratio/max": 1.2088912725448608, "entropy": 0.05566363618709147, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.0020491802133619785, "reward_total_mean": 0.9981216788291931, "reward_meter_mean": 0.9981216788291931, "reward_meter_std": 0.000908432062715292, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9981216788291931, "reward_total_composite_std": 0.000908432062715292, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 871.0} {"timestamp_utc": "2026-04-11T21:16:46Z", "mode": "train", "global_step": 872, "epoch": 0.03367315415508187, "loss": -0.0045, "grad_norm": 3.307403326034546, "learning_rate": 7.360606060606061e-06, "num_tokens": 1882747.0, "completions/mean_length": 50.0, "completions/min_length": 49.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9919772744178772, "rewards/meter/std": 0.00023237457207869738, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9919772744178772, "rewards/total_composite/std": 0.00023237457207869738, "reward": 0.9919772744178772, "reward_std": 0.00023237054119817913, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0074974666349589825, "sampling/sampling_logp_difference/max": 1.158355951309204, "sampling/importance_sampling_ratio/min": 0.31400200724601746, "sampling/importance_sampling_ratio/mean": 1.0007926225662231, "sampling/importance_sampling_ratio/max": 1.241955041885376, "entropy": 0.02604396513197571, "clip_ratio/low_mean": 0.0024999999441206455, "clip_ratio/low_min": 0.0024999999441206455, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0024999999441206455, "reward_total_mean": 0.9919772744178772, "reward_meter_mean": 0.9919772744178772, "reward_meter_std": 0.00023237457207869738, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9919772744178772, "reward_total_composite_std": 0.00023237457207869738, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 872.0} {"timestamp_utc": "2026-04-11T21:16:51Z", "mode": "train", "global_step": 873, "epoch": 0.03371177015755329, "loss": 0.0063, "grad_norm": 4.152334213256836, "learning_rate": 7.357575757575758e-06, "num_tokens": 1884839.0, "completions/mean_length": 95.5, "completions/min_length": 94.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.5, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9687960147857666, "rewards/meter/std": 0.012223567813634872, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9687960147857666, "rewards/total_composite/std": 0.012223567813634872, "reward": 0.9687960147857666, "reward_std": 0.012223553843796253, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023430444300174713, "sampling/sampling_logp_difference/max": 3.1329636573791504, "sampling/importance_sampling_ratio/min": 0.04358842223882675, "sampling/importance_sampling_ratio/mean": 0.9993469715118408, "sampling/importance_sampling_ratio/max": 1.6694515943527222, "entropy": 0.07936409767717123, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.011842688778415322, "clip_ratio/high_max": 0.011842688778415322, "clip_ratio/region_mean": 0.015748938778415322, "reward_total_mean": 0.9687960147857666, "reward_meter_mean": 0.9687960147857666, "reward_meter_std": 0.012223567813634872, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9687960147857666, "reward_total_composite_std": 0.012223567813634872, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 873.0} {"timestamp_utc": "2026-04-11T21:16:55Z", "mode": "train", "global_step": 874, "epoch": 0.033750386160024716, "loss": 0.0316, "grad_norm": 5.913609504699707, "learning_rate": 7.354545454545456e-06, "num_tokens": 1886388.0, "completions/mean_length": 35.625, "completions/min_length": 32.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.971309244632721, "rewards/meter/std": 0.027311865240335464, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.971309244632721, "rewards/total_composite/std": 0.027311865240335464, "reward": 0.971309244632721, "reward_std": 0.02731187455356121, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028721056878566742, "sampling/sampling_logp_difference/max": 0.9919209480285645, "sampling/importance_sampling_ratio/min": 0.37086358666419983, "sampling/importance_sampling_ratio/mean": 1.0020819902420044, "sampling/importance_sampling_ratio/max": 1.5368764400482178, "entropy": 0.12803277699276805, "clip_ratio/low_mean": 0.011029412038624287, "clip_ratio/low_min": 0.011029412038624287, "clip_ratio/high_mean": 0.018208998255431652, "clip_ratio/high_max": 0.018208998255431652, "clip_ratio/region_mean": 0.02923841029405594, "reward_total_mean": 0.971309244632721, "reward_meter_mean": 0.971309244632721, "reward_meter_std": 0.027311865240335464, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.971309244632721, "reward_total_composite_std": 0.027311865240335464, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 874.0} {"timestamp_utc": "2026-04-11T21:17:01Z", "mode": "train", "global_step": 875, "epoch": 0.03378900216249614, "loss": 0.0089, "grad_norm": 2.007058620452881, "learning_rate": 7.351515151515151e-06, "num_tokens": 1888918.0, "completions/mean_length": 155.25, "completions/min_length": 149.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 155.25, "completions/min_terminated_length": 149.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.624158501625061, "rewards/meter/std": 0.17649312317371368, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.624158501625061, "rewards/total_composite/std": 0.17649312317371368, "reward": 0.624158501625061, "reward_std": 0.17649312317371368, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018157916143536568, "sampling/sampling_logp_difference/max": 1.0785086154937744, "sampling/importance_sampling_ratio/min": 0.3401023745536804, "sampling/importance_sampling_ratio/mean": 1.0061261653900146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13584541296586394, "clip_ratio/low_mean": 0.00631756178336218, "clip_ratio/low_min": 0.00631756178336218, "clip_ratio/high_mean": 0.00903262326028198, "clip_ratio/high_max": 0.00903262326028198, "clip_ratio/region_mean": 0.01535018504364416, "reward_total_mean": 0.624158501625061, "reward_meter_mean": 0.624158501625061, "reward_meter_std": 0.17649312317371368, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.624158501625061, "reward_total_composite_std": 0.17649312317371368, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 875.0} {"timestamp_utc": "2026-04-11T21:17:06Z", "mode": "train", "global_step": 876, "epoch": 0.033827618164967564, "loss": -0.0151, "grad_norm": 7.30751371383667, "learning_rate": 7.348484848484849e-06, "num_tokens": 1890588.0, "completions/mean_length": 52.75, "completions/min_length": 51.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.75, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9927805662155151, "rewards/meter/std": 0.0017144550802186131, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9927805662155151, "rewards/total_composite/std": 0.0017144550802186131, "reward": 0.9927805662155151, "reward_std": 0.0017144549638032913, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010966307483613491, "sampling/sampling_logp_difference/max": 0.5893580317497253, "sampling/importance_sampling_ratio/min": 0.554683268070221, "sampling/importance_sampling_ratio/mean": 0.9988293647766113, "sampling/importance_sampling_ratio/max": 1.3743891716003418, "entropy": 0.04893670417368412, "clip_ratio/low_mean": 0.007167961681261659, "clip_ratio/low_min": 0.007167961681261659, "clip_ratio/high_mean": 0.006861772621050477, "clip_ratio/high_max": 0.006861772621050477, "clip_ratio/region_mean": 0.014029734302312136, "reward_total_mean": 0.9927805662155151, "reward_meter_mean": 0.9927805662155151, "reward_meter_std": 0.0017144550802186131, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9927805662155151, "reward_total_composite_std": 0.0017144550802186131, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 876.0} {"timestamp_utc": "2026-04-11T21:17:12Z", "mode": "train", "global_step": 877, "epoch": 0.03386623416743899, "loss": -0.0002, "grad_norm": 0.5615001916885376, "learning_rate": 7.345454545454546e-06, "num_tokens": 1893308.0, "completions/mean_length": 151.0, "completions/min_length": 151.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.0, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.9985967874526978, "rewards/meter/std": 0.000119388809252996, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985967874526978, "rewards/total_composite/std": 0.000119388809252996, "reward": 0.9985967874526978, "reward_std": 0.00011939799878746271, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0013662419514730573, "sampling/sampling_logp_difference/max": 0.3626127243041992, "sampling/importance_sampling_ratio/min": 0.7126171588897705, "sampling/importance_sampling_ratio/mean": 1.0003682374954224, "sampling/importance_sampling_ratio/max": 1.4370793104171753, "entropy": 0.008502666431013495, "clip_ratio/low_mean": 0.0008278145687654614, "clip_ratio/low_min": 0.0008278145687654614, "clip_ratio/high_mean": 0.0008278145687654614, "clip_ratio/high_max": 0.0008278145687654614, "clip_ratio/region_mean": 0.0016556291375309229, "reward_total_mean": 0.9985967874526978, "reward_meter_mean": 0.9985967874526978, "reward_meter_std": 0.000119388809252996, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985967874526978, "reward_total_composite_std": 0.000119388809252996, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 877.0} {"timestamp_utc": "2026-04-11T21:17:17Z", "mode": "train", "global_step": 878, "epoch": 0.03390485016991041, "loss": 0.0021, "grad_norm": 3.876832962036133, "learning_rate": 7.342424242424243e-06, "num_tokens": 1895197.0, "completions/mean_length": 74.125, "completions/min_length": 73.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.125, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9949108362197876, "rewards/meter/std": 0.0007706816541031003, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9949108362197876, "rewards/total_composite/std": 0.0007706816541031003, "reward": 0.9949108362197876, "reward_std": 0.0007706801407039165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013677185401320457, "sampling/sampling_logp_difference/max": 2.7378652095794678, "sampling/importance_sampling_ratio/min": 0.06470833718776703, "sampling/importance_sampling_ratio/mean": 0.9976504445075989, "sampling/importance_sampling_ratio/max": 1.3579704761505127, "entropy": 0.03811599453911185, "clip_ratio/low_mean": 0.0016891892300918698, "clip_ratio/low_min": 0.0016891892300918698, "clip_ratio/high_mean": 0.005000000121071935, "clip_ratio/high_max": 0.005000000121071935, "clip_ratio/region_mean": 0.0066891893511638045, "reward_total_mean": 0.9949108362197876, "reward_meter_mean": 0.9949108362197876, "reward_meter_std": 0.0007706816541031003, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9949108362197876, "reward_total_composite_std": 0.0007706816541031003, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 878.0} {"timestamp_utc": "2026-04-11T21:17:27Z", "mode": "train", "global_step": 879, "epoch": 0.033943466172381836, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.3393939393939395e-06, "num_tokens": 1896837.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9602138996124268, "rewards/meter/std": 0.10911035537719727, "rewards/count_adherence/mean": 0.8611111044883728, "rewards/count_adherence/std": 0.2357022613286972, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8253891468048096, "rewards/total_composite/std": 0.33350756764411926, "reward": 0.8253891468048096, "reward_std": 0.33350756764411926, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8253891468048096, "reward_meter_mean": 0.9602138996124268, "reward_meter_std": 0.10911035537719727, "reward_count_adherence_mean": 0.8611111044883728, "reward_count_adherence_std": 0.2357022613286972, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8253891468048096, "reward_total_composite_std": 0.33350756764411926, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 879.0} {"timestamp_utc": "2026-04-11T21:17:33Z", "mode": "train", "global_step": 880, "epoch": 0.03398208217485326, "loss": -0.0271, "grad_norm": 3.2072367668151855, "learning_rate": 7.336363636363637e-06, "num_tokens": 1899715.0, "completions/mean_length": 149.75, "completions/min_length": 140.0, "completions/max_length": 168.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.75, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 168.0, "rewards/meter/mean": 0.20864656567573547, "rewards/meter/std": 0.1737600564956665, "rewards/count_adherence/mean": 0.8928571939468384, "rewards/count_adherence/std": 0.06613000482320786, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.19013437628746033, "rewards/total_composite/std": 0.17502154409885406, "reward": 0.19013437628746033, "reward_std": 0.17502154409885406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01582314446568489, "sampling/sampling_logp_difference/max": 1.8117260932922363, "sampling/importance_sampling_ratio/min": 0.16337189078330994, "sampling/importance_sampling_ratio/mean": 0.998909056186676, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03868780063930899, "clip_ratio/low_mean": 0.006697957520373166, "clip_ratio/low_min": 0.006697957520373166, "clip_ratio/high_mean": 0.004464285913854837, "clip_ratio/high_max": 0.004464285913854837, "clip_ratio/region_mean": 0.011162243434228003, "reward_total_mean": 0.19013437628746033, "reward_meter_mean": 0.20864656567573547, "reward_meter_std": 0.1737600564956665, "reward_count_adherence_mean": 0.8928571939468384, "reward_count_adherence_std": 0.06613000482320786, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.19013437628746033, "reward_total_composite_std": 0.17502154409885406, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 880.0} {"timestamp_utc": "2026-04-11T21:17:38Z", "mode": "train", "global_step": 881, "epoch": 0.034020698177324685, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.333333333333333e-06, "num_tokens": 1901491.0, "completions/mean_length": 46.0, "completions/min_length": 46.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9776018261909485, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9776018261909485, "rewards/total_composite/std": 0.0, "reward": 0.9776018261909485, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.001749740680679679, "sampling/sampling_logp_difference/max": 0.06094535440206528, "sampling/importance_sampling_ratio/min": 0.9408746957778931, "sampling/importance_sampling_ratio/mean": 1.0010552406311035, "sampling/importance_sampling_ratio/max": 1.057377815246582, "entropy": 0.013820012216456234, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9776018261909485, "reward_meter_mean": 0.9776018261909485, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9776018261909485, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 881.0} {"timestamp_utc": "2026-04-11T21:17:43Z", "mode": "train", "global_step": 882, "epoch": 0.03405931417979611, "loss": -0.0104, "grad_norm": 4.600780963897705, "learning_rate": 7.330303030303031e-06, "num_tokens": 1903216.0, "completions/mean_length": 67.625, "completions/min_length": 66.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.846278965473175, "rewards/meter/std": 0.23601077497005463, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.846278965473175, "rewards/total_composite/std": 0.23601077497005463, "reward": 0.846278965473175, "reward_std": 0.23601076006889343, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007855754345655441, "sampling/sampling_logp_difference/max": 0.9119863510131836, "sampling/importance_sampling_ratio/min": 0.4017254710197449, "sampling/importance_sampling_ratio/mean": 1.0006687641143799, "sampling/importance_sampling_ratio/max": 1.4821372032165527, "entropy": 0.03605632018297911, "clip_ratio/low_mean": 0.005681818351149559, "clip_ratio/low_min": 0.005681818351149559, "clip_ratio/high_mean": 0.0036764706019312143, "clip_ratio/high_max": 0.0036764706019312143, "clip_ratio/region_mean": 0.009358288953080773, "reward_total_mean": 0.846278965473175, "reward_meter_mean": 0.846278965473175, "reward_meter_std": 0.23601077497005463, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.846278965473175, "reward_total_composite_std": 0.23601077497005463, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 882.0} {"timestamp_utc": "2026-04-11T21:17:49Z", "mode": "train", "global_step": 883, "epoch": 0.03409793018226753, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.3272727272727285e-06, "num_tokens": 1907008.0, "completions/mean_length": 211.0, "completions/min_length": 211.0, "completions/max_length": 211.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 211.0, "completions/min_terminated_length": 211.0, "completions/max_terminated_length": 211.0, "rewards/meter/mean": 0.9987902045249939, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9987902045249939, "rewards/total_composite/std": 0.0, "reward": 0.9987902045249939, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006794043583795428, "sampling/sampling_logp_difference/max": 0.23176756501197815, "sampling/importance_sampling_ratio/min": 0.7931305170059204, "sampling/importance_sampling_ratio/mean": 0.9999943971633911, "sampling/importance_sampling_ratio/max": 1.0602232217788696, "entropy": 0.004078510770341381, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9987902045249939, "reward_meter_mean": 0.9987902045249939, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9987902045249939, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 883.0} {"timestamp_utc": "2026-04-11T21:17:54Z", "mode": "train", "global_step": 884, "epoch": 0.03413654618473896, "loss": -0.0068, "grad_norm": 2.7239935398101807, "learning_rate": 7.324242424242425e-06, "num_tokens": 1908515.0, "completions/mean_length": 34.375, "completions/min_length": 34.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9869133234024048, "rewards/meter/std": 0.003022974357008934, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9869133234024048, "rewards/total_composite/std": 0.003022974357008934, "reward": 0.9869133234024048, "reward_std": 0.003022971097379923, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020580653101205826, "sampling/sampling_logp_difference/max": 0.7628744840621948, "sampling/importance_sampling_ratio/min": 0.4663240611553192, "sampling/importance_sampling_ratio/mean": 0.998308539390564, "sampling/importance_sampling_ratio/max": 1.3481626510620117, "entropy": 0.06947248196229339, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/high_mean": 0.02542016818188131, "clip_ratio/high_max": 0.02542016818188131, "clip_ratio/region_mean": 0.029096638783812523, "reward_total_mean": 0.9869133234024048, "reward_meter_mean": 0.9869133234024048, "reward_meter_std": 0.003022974357008934, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9869133234024048, "reward_total_composite_std": 0.003022974357008934, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 884.0} {"timestamp_utc": "2026-04-11T21:17:59Z", "mode": "train", "global_step": 885, "epoch": 0.03417516218721038, "loss": 0.0112, "grad_norm": 2.5803024768829346, "learning_rate": 7.321212121212122e-06, "num_tokens": 1910591.0, "completions/mean_length": 77.5, "completions/min_length": 74.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9319126605987549, "rewards/meter/std": 0.12844786047935486, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9319126605987549, "rewards/total_composite/std": 0.12844786047935486, "reward": 0.9319126605987549, "reward_std": 0.12844786047935486, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01291993074119091, "sampling/sampling_logp_difference/max": 0.9233436584472656, "sampling/importance_sampling_ratio/min": 0.3971887528896332, "sampling/importance_sampling_ratio/mean": 0.9978067278862, "sampling/importance_sampling_ratio/max": 1.3109866380691528, "entropy": 0.07723345141857862, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/high_mean": 0.009830259950831532, "clip_ratio/high_max": 0.009830259950831532, "clip_ratio/region_mean": 0.011392759974114597, "reward_total_mean": 0.9319126605987549, "reward_meter_mean": 0.9319126605987549, "reward_meter_std": 0.12844786047935486, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9319126605987549, "reward_total_composite_std": 0.12844786047935486, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 885.0} {"timestamp_utc": "2026-04-11T21:18:03Z", "mode": "train", "global_step": 886, "epoch": 0.034213778189681805, "loss": -0.0326, "grad_norm": 9.5073823928833, "learning_rate": 7.3181818181818186e-06, "num_tokens": 1912563.0, "completions/mean_length": 86.5, "completions/min_length": 78.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.5, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.3333978056907654, "rewards/meter/std": 0.34304332733154297, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3333978056907654, "rewards/total_composite/std": 0.34304332733154297, "reward": 0.3333978056907654, "reward_std": 0.34304332733154297, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04321261867880821, "sampling/sampling_logp_difference/max": 2.5233585834503174, "sampling/importance_sampling_ratio/min": 0.10055816173553467, "sampling/importance_sampling_ratio/mean": 1.0085729360580444, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.187221460044384, "clip_ratio/low_mean": 0.021021681604906917, "clip_ratio/low_min": 0.021021681604906917, "clip_ratio/high_mean": 0.005561735248193145, "clip_ratio/high_max": 0.005561735248193145, "clip_ratio/region_mean": 0.02658341685310006, "reward_total_mean": 0.3333978056907654, "reward_meter_mean": 0.3333978056907654, "reward_meter_std": 0.34304332733154297, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3333978056907654, "reward_total_composite_std": 0.34304332733154297, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 886.0} {"timestamp_utc": "2026-04-11T21:18:08Z", "mode": "train", "global_step": 887, "epoch": 0.03425239419215323, "loss": 0.0007, "grad_norm": 3.4898829460144043, "learning_rate": 7.315151515151516e-06, "num_tokens": 1914302.0, "completions/mean_length": 55.375, "completions/min_length": 54.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.375, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9809094667434692, "rewards/meter/std": 0.010828851722180843, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9809094667434692, "rewards/total_composite/std": 0.010828851722180843, "reward": 0.9809094667434692, "reward_std": 0.010828849859535694, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014852515421807766, "sampling/sampling_logp_difference/max": 0.9145876169204712, "sampling/importance_sampling_ratio/min": 0.40068185329437256, "sampling/importance_sampling_ratio/mean": 1.0075335502624512, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06546596344560385, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.008974857395514846, "clip_ratio/high_max": 0.008974857395514846, "clip_ratio/region_mean": 0.008974857395514846, "reward_total_mean": 0.9809094667434692, "reward_meter_mean": 0.9809094667434692, "reward_meter_std": 0.010828851722180843, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9809094667434692, "reward_total_composite_std": 0.010828851722180843, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 887.0} {"timestamp_utc": "2026-04-11T21:18:13Z", "mode": "train", "global_step": 888, "epoch": 0.03429101019462465, "loss": 0.0105, "grad_norm": 11.404142379760742, "learning_rate": 7.312121212121212e-06, "num_tokens": 1916038.0, "completions/mean_length": 37.0, "completions/min_length": 36.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9462078809738159, "rewards/meter/std": 0.029821377247571945, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9462078809738159, "rewards/total_composite/std": 0.029821377247571945, "reward": 0.9462078809738159, "reward_std": 0.0298213679343462, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018947336822748184, "sampling/sampling_logp_difference/max": 0.6967822313308716, "sampling/importance_sampling_ratio/min": 0.49818578362464905, "sampling/importance_sampling_ratio/mean": 0.99570631980896, "sampling/importance_sampling_ratio/max": 1.390830636024475, "entropy": 0.07240521302446723, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.01979166711680591, "clip_ratio/high_max": 0.01979166711680591, "clip_ratio/region_mean": 0.01979166711680591, "reward_total_mean": 0.9462078809738159, "reward_meter_mean": 0.9462078809738159, "reward_meter_std": 0.029821377247571945, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9462078809738159, "reward_total_composite_std": 0.029821377247571945, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 888.0} {"timestamp_utc": "2026-04-11T21:18:17Z", "mode": "train", "global_step": 889, "epoch": 0.03432962619709608, "loss": 0.0655, "grad_norm": 26.710201263427734, "learning_rate": 7.30909090909091e-06, "num_tokens": 1917652.0, "completions/mean_length": 36.75, "completions/min_length": 36.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9549926519393921, "rewards/meter/std": 0.00983886606991291, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9549926519393921, "rewards/total_composite/std": 0.00983886606991291, "reward": 0.9549926519393921, "reward_std": 0.009838856756687164, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005750642623752356, "sampling/sampling_logp_difference/max": 0.3002748489379883, "sampling/importance_sampling_ratio/min": 0.8049483299255371, "sampling/importance_sampling_ratio/mean": 1.001970648765564, "sampling/importance_sampling_ratio/max": 1.3502299785614014, "entropy": 0.029969313880428672, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0034722222480922937, "clip_ratio/high_max": 0.0034722222480922937, "clip_ratio/region_mean": 0.0034722222480922937, "reward_total_mean": 0.9549926519393921, "reward_meter_mean": 0.9549926519393921, "reward_meter_std": 0.00983886606991291, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9549926519393921, "reward_total_composite_std": 0.00983886606991291, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 889.0} {"timestamp_utc": "2026-04-11T21:18:22Z", "mode": "train", "global_step": 890, "epoch": 0.0343682421995675, "loss": 0.0454, "grad_norm": 16.860692977905273, "learning_rate": 7.306060606060607e-06, "num_tokens": 1919315.0, "completions/mean_length": 42.875, "completions/min_length": 42.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.875, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.18164604902267456, "rewards/meter/std": 0.044547002762556076, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.18164604902267456, "rewards/total_composite/std": 0.044547002762556076, "reward": 0.18164604902267456, "reward_std": 0.044547002762556076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019292106851935387, "sampling/sampling_logp_difference/max": 1.5946221351623535, "sampling/importance_sampling_ratio/min": 0.20298521220684052, "sampling/importance_sampling_ratio/mean": 0.997443675994873, "sampling/importance_sampling_ratio/max": 1.6797680854797363, "entropy": 0.06942054536193609, "clip_ratio/low_mean": 0.010204081423580647, "clip_ratio/low_min": 0.010204081423580647, "clip_ratio/high_mean": 0.0029761905316263437, "clip_ratio/high_max": 0.0029761905316263437, "clip_ratio/region_mean": 0.01318027195520699, "reward_total_mean": 0.18164604902267456, "reward_meter_mean": 0.18164604902267456, "reward_meter_std": 0.044547002762556076, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.18164604902267456, "reward_total_composite_std": 0.044547002762556076, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 890.0} {"timestamp_utc": "2026-04-11T21:18:27Z", "mode": "train", "global_step": 891, "epoch": 0.034406858202038926, "loss": -0.0053, "grad_norm": 11.439748764038086, "learning_rate": 7.303030303030304e-06, "num_tokens": 1921169.0, "completions/mean_length": 66.75, "completions/min_length": 65.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.75, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9968581199645996, "rewards/meter/std": 0.00707624526694417, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9968581199645996, "rewards/total_composite/std": 0.00707624526694417, "reward": 0.9968581199645996, "reward_std": 0.007076251320540905, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003642734605818987, "sampling/sampling_logp_difference/max": 0.22027695178985596, "sampling/importance_sampling_ratio/min": 0.8022965788841248, "sampling/importance_sampling_ratio/mean": 1.0001572370529175, "sampling/importance_sampling_ratio/max": 1.1490119695663452, "entropy": 0.02447754261083901, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005597014795057476, "clip_ratio/high_max": 0.005597014795057476, "clip_ratio/region_mean": 0.005597014795057476, "reward_total_mean": 0.9968581199645996, "reward_meter_mean": 0.9968581199645996, "reward_meter_std": 0.00707624526694417, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9968581199645996, "reward_total_composite_std": 0.00707624526694417, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 891.0} {"timestamp_utc": "2026-04-11T21:18:32Z", "mode": "train", "global_step": 892, "epoch": 0.03444547420451035, "loss": -0.0002, "grad_norm": 3.7469489574432373, "learning_rate": 7.3e-06, "num_tokens": 1923148.0, "completions/mean_length": 82.375, "completions/min_length": 78.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.375, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9940004348754883, "rewards/meter/std": 0.002209783997386694, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9940004348754883, "rewards/total_composite/std": 0.002209783997386694, "reward": 0.9940004348754883, "reward_std": 0.0022097795736044645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038082949817180634, "sampling/sampling_logp_difference/max": 6.6528401374816895, "sampling/importance_sampling_ratio/min": 0.0012903522001579404, "sampling/importance_sampling_ratio/mean": 0.9966236352920532, "sampling/importance_sampling_ratio/max": 1.4095396995544434, "entropy": 0.08194271847605705, "clip_ratio/low_mean": 0.02002646797336638, "clip_ratio/low_min": 0.02002646797336638, "clip_ratio/high_mean": 0.009072876535356045, "clip_ratio/high_max": 0.009072876535356045, "clip_ratio/region_mean": 0.029099344508722425, "reward_total_mean": 0.9940004348754883, "reward_meter_mean": 0.9940004348754883, "reward_meter_std": 0.002209783997386694, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9940004348754883, "reward_total_composite_std": 0.002209783997386694, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 892.0} {"timestamp_utc": "2026-04-11T21:18:37Z", "mode": "train", "global_step": 893, "epoch": 0.034484090206981774, "loss": 0.0053, "grad_norm": 5.554684638977051, "learning_rate": 7.296969696969698e-06, "num_tokens": 1924778.0, "completions/mean_length": 34.75, "completions/min_length": 33.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9837242364883423, "rewards/meter/std": 0.0036802212707698345, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9837242364883423, "rewards/total_composite/std": 0.0036802212707698345, "reward": 0.9837242364883423, "reward_std": 0.003680221037939191, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022954892367124557, "sampling/sampling_logp_difference/max": 1.4423182010650635, "sampling/importance_sampling_ratio/min": 0.27617737650871277, "sampling/importance_sampling_ratio/mean": 1.0071067810058594, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06711644260212779, "clip_ratio/low_mean": 0.01093073608353734, "clip_ratio/low_min": 0.01093073608353734, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/region_mean": 0.014502164674922824, "reward_total_mean": 0.9837242364883423, "reward_meter_mean": 0.9837242364883423, "reward_meter_std": 0.0036802212707698345, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9837242364883423, "reward_total_composite_std": 0.0036802212707698345, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 893.0} {"timestamp_utc": "2026-04-11T21:18:43Z", "mode": "train", "global_step": 894, "epoch": 0.0345227062094532, "loss": -0.0169, "grad_norm": 1.8849190473556519, "learning_rate": 7.293939393939394e-06, "num_tokens": 1927540.0, "completions/mean_length": 158.25, "completions/min_length": 142.0, "completions/max_length": 171.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.25, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.9966323375701904, "rewards/meter/std": 0.0018326297868043184, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.971698522567749, "rewards/total_composite/std": 0.07025559991598129, "reward": 0.971698522567749, "reward_std": 0.0702555924654007, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007320974953472614, "sampling/sampling_logp_difference/max": 0.905490517616272, "sampling/importance_sampling_ratio/min": 0.40434351563453674, "sampling/importance_sampling_ratio/mean": 1.000131368637085, "sampling/importance_sampling_ratio/max": 1.72441565990448, "entropy": 0.038358791498467326, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007113803003448993, "clip_ratio/high_max": 0.007113803003448993, "clip_ratio/region_mean": 0.007113803003448993, "reward_total_mean": 0.971698522567749, "reward_meter_mean": 0.9966323375701904, "reward_meter_std": 0.0018326297868043184, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.971698522567749, "reward_total_composite_std": 0.07025559991598129, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 894.0} {"timestamp_utc": "2026-04-11T21:18:47Z", "mode": "train", "global_step": 895, "epoch": 0.03456132221192462, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.290909090909092e-06, "num_tokens": 1929356.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9993599653244019, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9993599653244019, "rewards/total_composite/std": 0.0, "reward": 0.9993599653244019, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0008930930052883923, "sampling/sampling_logp_difference/max": 0.036980029195547104, "sampling/importance_sampling_ratio/min": 0.9873042702674866, "sampling/importance_sampling_ratio/mean": 1.0008255243301392, "sampling/importance_sampling_ratio/max": 1.0376722812652588, "entropy": 0.008365467539988458, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993599653244019, "reward_meter_mean": 0.9993599653244019, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9993599653244019, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 895.0} {"timestamp_utc": "2026-04-11T21:18:52Z", "mode": "train", "global_step": 896, "epoch": 0.034599938214396046, "loss": -0.0002, "grad_norm": 0.5587096214294434, "learning_rate": 7.287878787878789e-06, "num_tokens": 1931212.0, "completions/mean_length": 71.0, "completions/min_length": 71.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9734580516815186, "rewards/meter/std": 0.0019603946711868048, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9734580516815186, "rewards/total_composite/std": 0.0019603946711868048, "reward": 0.9734580516815186, "reward_std": 0.0019604084081947803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005342015065252781, "sampling/sampling_logp_difference/max": 0.5951647758483887, "sampling/importance_sampling_ratio/min": 0.5514717102050781, "sampling/importance_sampling_ratio/mean": 1.0023491382598877, "sampling/importance_sampling_ratio/max": 1.1052722930908203, "entropy": 0.039308680687099695, "clip_ratio/low_mean": 0.0017605633474886417, "clip_ratio/low_min": 0.0017605633474886417, "clip_ratio/high_mean": 0.0017605633474886417, "clip_ratio/high_max": 0.0017605633474886417, "clip_ratio/region_mean": 0.0035211266949772835, "reward_total_mean": 0.9734580516815186, "reward_meter_mean": 0.9734580516815186, "reward_meter_std": 0.0019603946711868048, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9734580516815186, "reward_total_composite_std": 0.0019603946711868048, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 896.0} {"timestamp_utc": "2026-04-11T21:18:57Z", "mode": "train", "global_step": 897, "epoch": 0.03463855421686747, "loss": 0.0001, "grad_norm": 0.1974111944437027, "learning_rate": 7.284848484848486e-06, "num_tokens": 1933076.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9983253479003906, "rewards/meter/std": 0.00010776949056889862, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9983253479003906, "rewards/total_composite/std": 0.00010776949056889862, "reward": 0.9983253479003906, "reward_std": 0.00010777551506180316, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0014991178177297115, "sampling/sampling_logp_difference/max": 0.23564481735229492, "sampling/importance_sampling_ratio/min": 0.7900612354278564, "sampling/importance_sampling_ratio/mean": 1.0002027750015259, "sampling/importance_sampling_ratio/max": 1.0662466287612915, "entropy": 0.011938330717384815, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.0020491802133619785, "reward_total_mean": 0.9983253479003906, "reward_meter_mean": 0.9983253479003906, "reward_meter_std": 0.00010776949056889862, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9983253479003906, "reward_total_composite_std": 0.00010776949056889862, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 897.0} {"timestamp_utc": "2026-04-11T21:19:02Z", "mode": "train", "global_step": 898, "epoch": 0.034677170219338894, "loss": -0.0461, "grad_norm": 6.410183429718018, "learning_rate": 7.281818181818182e-06, "num_tokens": 1935130.0, "completions/mean_length": 91.75, "completions/min_length": 85.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.75, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9361319541931152, "rewards/meter/std": 0.09528474509716034, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9361319541931152, "rewards/total_composite/std": 0.09528474509716034, "reward": 0.9361319541931152, "reward_std": 0.09528473019599915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017373455688357353, "sampling/sampling_logp_difference/max": 2.314580202102661, "sampling/importance_sampling_ratio/min": 0.09880765527486801, "sampling/importance_sampling_ratio/mean": 1.0027812719345093, "sampling/importance_sampling_ratio/max": 1.8476688861846924, "entropy": 0.049219525419175625, "clip_ratio/low_mean": 0.008823529700748622, "clip_ratio/low_min": 0.008823529700748622, "clip_ratio/high_mean": 0.005249504814855754, "clip_ratio/high_max": 0.005249504814855754, "clip_ratio/region_mean": 0.014073034515604377, "reward_total_mean": 0.9361319541931152, "reward_meter_mean": 0.9361319541931152, "reward_meter_std": 0.09528474509716034, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9361319541931152, "reward_total_composite_std": 0.09528474509716034, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 898.0} {"timestamp_utc": "2026-04-11T21:19:06Z", "mode": "train", "global_step": 899, "epoch": 0.03471578622181032, "loss": -0.0078, "grad_norm": 7.529623508453369, "learning_rate": 7.2787878787878795e-06, "num_tokens": 1937070.0, "completions/mean_length": 64.5, "completions/min_length": 58.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8412382006645203, "rewards/meter/std": 0.3296927809715271, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8412382006645203, "rewards/total_composite/std": 0.3296927809715271, "reward": 0.8412382006645203, "reward_std": 0.3296927809715271, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01657470501959324, "sampling/sampling_logp_difference/max": 0.8382200002670288, "sampling/importance_sampling_ratio/min": 0.4324796497821808, "sampling/importance_sampling_ratio/mean": 0.9965075254440308, "sampling/importance_sampling_ratio/max": 1.465029001235962, "entropy": 0.06669788854196668, "clip_ratio/low_mean": 0.0021551724057644606, "clip_ratio/low_min": 0.0021551724057644606, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.0041082974057644606, "reward_total_mean": 0.8412382006645203, "reward_meter_mean": 0.8412382006645203, "reward_meter_std": 0.3296927809715271, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8412382006645203, "reward_total_composite_std": 0.3296927809715271, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 899.0} {"timestamp_utc": "2026-04-11T21:19:11Z", "mode": "train", "global_step": 900, "epoch": 0.03475440222428174, "loss": -0.0017, "grad_norm": 5.273963451385498, "learning_rate": 7.275757575757576e-06, "num_tokens": 1938662.0, "completions/mean_length": 46.0, "completions/min_length": 46.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9821657538414001, "rewards/meter/std": 0.0028169082943350077, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9821657538414001, "rewards/total_composite/std": 0.0028169082943350077, "reward": 0.9821657538414001, "reward_std": 0.0028168989811092615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006294311489909887, "sampling/sampling_logp_difference/max": 0.4679220914840698, "sampling/importance_sampling_ratio/min": 0.6263023018836975, "sampling/importance_sampling_ratio/mean": 0.9979045987129211, "sampling/importance_sampling_ratio/max": 1.313886284828186, "entropy": 0.02442757459357381, "clip_ratio/low_mean": 0.0027173913549631834, "clip_ratio/low_min": 0.0027173913549631834, "clip_ratio/high_mean": 0.0027173913549631834, "clip_ratio/high_max": 0.0027173913549631834, "clip_ratio/region_mean": 0.005434782709926367, "reward_total_mean": 0.9821657538414001, "reward_meter_mean": 0.9821657538414001, "reward_meter_std": 0.0028169082943350077, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9821657538414001, "reward_total_composite_std": 0.0028169082943350077, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 900.0} {"timestamp_utc": "2026-04-11T21:20:34Z", "mode": "eval", "global_step": 900, "epoch": 0.03475440222428174, "eval_loss": NaN, "eval_runtime": 82.6076, "eval_samples_per_second": 1.259, "eval_steps_per_second": 0.157, "eval_num_tokens": 1938662.0, "eval_completions/mean_length": 210.06730769230768, "eval_completions/min_length": 53.84615384615385, "eval_completions/max_length": 440.9230769230769, "eval_completions/clipped_ratio": 0.04807692307692308, "eval_completions/mean_terminated_length": 194.37637681227463, "eval_completions/min_terminated_length": 53.84615384615385, "eval_completions/max_terminated_length": 410.38461538461536, "eval_rewards/meter/mean": 0.6910155094586886, "eval_rewards/meter/std": 0.410910251048895, "eval_rewards/count_adherence/mean": 0.9650719991097083, "eval_rewards/count_adherence/std": 0.06803592738623802, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.08158924258672275, "eval_rewards/total_composite/mean": 0.6678107655965365, "eval_rewards/total_composite/std": 0.4100706663269263, "eval_reward": 0.6678107655965365, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0035032882946185195, "eval_sampling/sampling_logp_difference/max": 0.8707265624633203, "eval_sampling/importance_sampling_ratio/min": 0.47329516823475176, "eval_sampling/importance_sampling_ratio/mean": 1.0005392523912282, "eval_sampling/importance_sampling_ratio/max": 1.388524082990793, "eval_entropy": 0.024077745870901987, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6678107655965365, "eval_reward_meter_mean": 0.6910155094586886, "eval_reward_meter_std": 0.410910251048895, "eval_reward_count_adherence_mean": 0.9650719991097083, "eval_reward_count_adherence_std": 0.06803592738623802, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.08158924258672275, "eval_reward_total_composite_mean": 0.6678107655965365, "eval_reward_total_composite_std": 0.4100706663269263, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 900.0} {"timestamp_utc": "2026-04-11T21:20:45Z", "mode": "train", "global_step": 901, "epoch": 0.03479301822675317, "loss": 0.0629, "grad_norm": 2.436157703399658, "learning_rate": 7.272727272727273e-06, "num_tokens": 1941909.0, "completions/mean_length": 208.875, "completions/min_length": 185.0, "completions/max_length": 239.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 208.875, "completions/min_terminated_length": 185.0, "completions/max_terminated_length": 239.0, "rewards/meter/mean": 0.8068044185638428, "rewards/meter/std": 0.3230624794960022, "rewards/count_adherence/mean": 0.9821428656578064, "rewards/count_adherence/std": 0.05050762742757797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7908886075019836, "rewards/total_composite/std": 0.3214382231235504, "reward": 0.7908886075019836, "reward_std": 0.3214382231235504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023887135088443756, "sampling/sampling_logp_difference/max": 3.419362783432007, "sampling/importance_sampling_ratio/min": 0.032733283936977386, "sampling/importance_sampling_ratio/mean": 1.00227689743042, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09786765673197806, "clip_ratio/low_mean": 0.008257969515398145, "clip_ratio/low_min": 0.008257969515398145, "clip_ratio/high_mean": 0.00740226759808138, "clip_ratio/high_max": 0.00740226759808138, "clip_ratio/region_mean": 0.015660237113479525, "reward_total_mean": 0.7908886075019836, "reward_meter_mean": 0.8068044185638428, "reward_meter_std": 0.3230624794960022, "reward_count_adherence_mean": 0.9821428656578064, "reward_count_adherence_std": 0.05050762742757797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7908886075019836, "reward_total_composite_std": 0.3214382231235504, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 901.0} {"timestamp_utc": "2026-04-11T21:20:49Z", "mode": "train", "global_step": 902, "epoch": 0.03483163422922459, "loss": 0.0016, "grad_norm": 3.1199753284454346, "learning_rate": 7.26969696969697e-06, "num_tokens": 1943558.0, "completions/mean_length": 46.125, "completions/min_length": 46.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9835736155509949, "rewards/meter/std": 0.00032080072560347617, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9835736155509949, "rewards/total_composite/std": 0.00032080072560347617, "reward": 0.9835736155509949, "reward_std": 0.0003208097768947482, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005055200308561325, "sampling/sampling_logp_difference/max": 1.011568546295166, "sampling/importance_sampling_ratio/min": 0.3636481463909149, "sampling/importance_sampling_ratio/mean": 0.9978182911872864, "sampling/importance_sampling_ratio/max": 1.0697872638702393, "entropy": 0.01709457952529192, "clip_ratio/low_mean": 0.002659574383869767, "clip_ratio/low_min": 0.002659574383869767, "clip_ratio/high_mean": 0.0027173913549631834, "clip_ratio/high_max": 0.0027173913549631834, "clip_ratio/region_mean": 0.005376965738832951, "reward_total_mean": 0.9835736155509949, "reward_meter_mean": 0.9835736155509949, "reward_meter_std": 0.00032080072560347617, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9835736155509949, "reward_total_composite_std": 0.00032080072560347617, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 902.0} {"timestamp_utc": "2026-04-11T21:20:54Z", "mode": "train", "global_step": 903, "epoch": 0.034870250231696015, "loss": 0.0227, "grad_norm": 7.262354850769043, "learning_rate": 7.266666666666668e-06, "num_tokens": 1945142.0, "completions/mean_length": 56.0, "completions/min_length": 55.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9269980192184448, "rewards/meter/std": 0.03096199780702591, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9269980192184448, "rewards/total_composite/std": 0.03096199780702591, "reward": 0.9269980192184448, "reward_std": 0.030961990356445312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03562089428305626, "sampling/sampling_logp_difference/max": 4.980681896209717, "sampling/importance_sampling_ratio/min": 0.006869376637041569, "sampling/importance_sampling_ratio/mean": 0.9985175132751465, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08563533145934343, "clip_ratio/low_mean": 0.011283891275525093, "clip_ratio/low_min": 0.011283891275525093, "clip_ratio/high_mean": 0.0022321429569274187, "clip_ratio/high_max": 0.0022321429569274187, "clip_ratio/region_mean": 0.013516034232452512, "reward_total_mean": 0.9269980192184448, "reward_meter_mean": 0.9269980192184448, "reward_meter_std": 0.03096199780702591, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9269980192184448, "reward_total_composite_std": 0.03096199780702591, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 903.0} {"timestamp_utc": "2026-04-11T21:21:00Z", "mode": "train", "global_step": 904, "epoch": 0.03490886623416744, "loss": 0.1366, "grad_norm": 3.5699656009674072, "learning_rate": 7.263636363636364e-06, "num_tokens": 1947415.0, "completions/mean_length": 113.125, "completions/min_length": 97.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.125, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9969273209571838, "rewards/meter/std": 7.628801540704444e-05, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8307713866233826, "rewards/total_composite/std": 0.1776193529367447, "reward": 0.8307713866233826, "reward_std": 0.1776193380355835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009861018508672714, "sampling/sampling_logp_difference/max": 0.7632689476013184, "sampling/importance_sampling_ratio/min": 0.46614015102386475, "sampling/importance_sampling_ratio/mean": 0.9973547458648682, "sampling/importance_sampling_ratio/max": 1.6411552429199219, "entropy": 0.03542015980929136, "clip_ratio/low_mean": 0.0029069767333567142, "clip_ratio/low_min": 0.0029069767333567142, "clip_ratio/high_mean": 0.0038659792626276612, "clip_ratio/high_max": 0.0038659792626276612, "clip_ratio/region_mean": 0.0067729559959843755, "reward_total_mean": 0.8307713866233826, "reward_meter_mean": 0.9969273209571838, "reward_meter_std": 7.628801540704444e-05, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8307713866233826, "reward_total_composite_std": 0.1776193529367447, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 904.0} {"timestamp_utc": "2026-04-11T21:21:05Z", "mode": "train", "global_step": 905, "epoch": 0.03494748223663886, "loss": 0.0307, "grad_norm": 8.248551368713379, "learning_rate": 7.260606060606061e-06, "num_tokens": 1948951.0, "completions/mean_length": 32.0, "completions/min_length": 31.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.7017570734024048, "rewards/meter/std": 0.1348530650138855, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7017570734024048, "rewards/total_composite/std": 0.1348530650138855, "reward": 0.7017570734024048, "reward_std": 0.1348530650138855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015220936387777328, "sampling/sampling_logp_difference/max": 0.5448028445243835, "sampling/importance_sampling_ratio/min": 0.5799561142921448, "sampling/importance_sampling_ratio/mean": 1.0027575492858887, "sampling/importance_sampling_ratio/max": 1.4159082174301147, "entropy": 0.08284427132457495, "clip_ratio/low_mean": 0.01171875, "clip_ratio/low_min": 0.01171875, "clip_ratio/high_mean": 0.012096773833036423, "clip_ratio/high_max": 0.012096773833036423, "clip_ratio/region_mean": 0.023815523833036423, "reward_total_mean": 0.7017570734024048, "reward_meter_mean": 0.7017570734024048, "reward_meter_std": 0.1348530650138855, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7017570734024048, "reward_total_composite_std": 0.1348530650138855, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 905.0} {"timestamp_utc": "2026-04-11T21:21:09Z", "mode": "train", "global_step": 906, "epoch": 0.03498609823911029, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.257575757575758e-06, "num_tokens": 1950375.0, "completions/mean_length": 23.0, "completions/min_length": 23.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.0, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.886601984500885, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.886601984500885, "rewards/total_composite/std": 0.0, "reward": 0.886601984500885, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0025429290253669024, "sampling/sampling_logp_difference/max": 0.09311866015195847, "sampling/importance_sampling_ratio/min": 0.9110854268074036, "sampling/importance_sampling_ratio/mean": 1.000812292098999, "sampling/importance_sampling_ratio/max": 1.034579873085022, "entropy": 0.01908092899248004, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.886601984500885, "reward_meter_mean": 0.886601984500885, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.886601984500885, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 906.0} {"timestamp_utc": "2026-04-11T21:21:13Z", "mode": "train", "global_step": 907, "epoch": 0.03502471424158171, "loss": -0.0144, "grad_norm": 7.115987300872803, "learning_rate": 7.254545454545455e-06, "num_tokens": 1952142.0, "completions/mean_length": 41.875, "completions/min_length": 40.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.1824103444814682, "rewards/meter/std": 0.0501907579600811, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.1824103444814682, "rewards/total_composite/std": 0.0501907579600811, "reward": 0.1824103444814682, "reward_std": 0.050190750509500504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01045476458966732, "sampling/sampling_logp_difference/max": 0.910797119140625, "sampling/importance_sampling_ratio/min": 0.502730667591095, "sampling/importance_sampling_ratio/mean": 1.0029710531234741, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03243764559738338, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005883167264983058, "clip_ratio/high_max": 0.005883167264983058, "clip_ratio/region_mean": 0.005883167264983058, "reward_total_mean": 0.1824103444814682, "reward_meter_mean": 0.1824103444814682, "reward_meter_std": 0.0501907579600811, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.1824103444814682, "reward_total_composite_std": 0.0501907579600811, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 907.0} {"timestamp_utc": "2026-04-11T21:21:18Z", "mode": "train", "global_step": 908, "epoch": 0.035063330244053136, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.251515151515151e-06, "num_tokens": 1954270.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00033485915628261864, "sampling/sampling_logp_difference/max": 0.03027336485683918, "sampling/importance_sampling_ratio/min": 0.9701802730560303, "sampling/importance_sampling_ratio/mean": 1.000209093093872, "sampling/importance_sampling_ratio/max": 1.0153448581695557, "entropy": 0.0028007474611513317, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 908.0} {"timestamp_utc": "2026-04-11T21:21:24Z", "mode": "train", "global_step": 909, "epoch": 0.03510194624652456, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.2484848484848495e-06, "num_tokens": 1955886.0, "completions/mean_length": 46.0, "completions/min_length": 46.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9836870431900024, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9836870431900024, "rewards/total_composite/std": 0.0, "reward": 0.9836870431900024, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0010412308620288968, "sampling/sampling_logp_difference/max": 0.1713903695344925, "sampling/importance_sampling_ratio/min": 0.8424926400184631, "sampling/importance_sampling_ratio/mean": 0.9999330043792725, "sampling/importance_sampling_ratio/max": 1.0079586505889893, "entropy": 0.0042151231318712234, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9836870431900024, "reward_meter_mean": 0.9836870431900024, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9836870431900024, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 909.0} {"timestamp_utc": "2026-04-11T21:21:29Z", "mode": "train", "global_step": 910, "epoch": 0.035140562248995984, "loss": 0.0589, "grad_norm": 7.724210262298584, "learning_rate": 7.245454545454546e-06, "num_tokens": 1957241.0, "completions/mean_length": 27.375, "completions/min_length": 26.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.375, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.8383634090423584, "rewards/meter/std": 0.28278204798698425, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8383634090423584, "rewards/total_composite/std": 0.28278204798698425, "reward": 0.8383634090423584, "reward_std": 0.28278201818466187, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03809576854109764, "sampling/sampling_logp_difference/max": 1.5702154636383057, "sampling/importance_sampling_ratio/min": 0.49254316091537476, "sampling/importance_sampling_ratio/mean": 1.0082052946090698, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15563157852739096, "clip_ratio/low_mean": 0.012500000651925802, "clip_ratio/low_min": 0.012500000651925802, "clip_ratio/high_mean": 0.019052707124501467, "clip_ratio/high_max": 0.019052707124501467, "clip_ratio/region_mean": 0.03155270777642727, "reward_total_mean": 0.8383634090423584, "reward_meter_mean": 0.8383634090423584, "reward_meter_std": 0.28278204798698425, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8383634090423584, "reward_total_composite_std": 0.28278204798698425, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 910.0} {"timestamp_utc": "2026-04-11T21:21:33Z", "mode": "train", "global_step": 911, "epoch": 0.03517917825146741, "loss": 0.0025, "grad_norm": 4.139067649841309, "learning_rate": 7.242424242424243e-06, "num_tokens": 1959073.0, "completions/mean_length": 69.0, "completions/min_length": 69.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8616424798965454, "rewards/meter/std": 0.012490366585552692, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8616424798965454, "rewards/total_composite/std": 0.012490366585552692, "reward": 0.8616424798965454, "reward_std": 0.012490374967455864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00576728954911232, "sampling/sampling_logp_difference/max": 1.4772365093231201, "sampling/importance_sampling_ratio/min": 0.22826765477657318, "sampling/importance_sampling_ratio/mean": 0.9986225366592407, "sampling/importance_sampling_ratio/max": 1.1270885467529297, "entropy": 0.012471764697693288, "clip_ratio/low_mean": 0.0018115942366421223, "clip_ratio/low_min": 0.0018115942366421223, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0018115942366421223, "reward_total_mean": 0.8616424798965454, "reward_meter_mean": 0.8616424798965454, "reward_meter_std": 0.012490366585552692, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8616424798965454, "reward_total_composite_std": 0.012490366585552692, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 911.0} {"timestamp_utc": "2026-04-11T21:21:38Z", "mode": "train", "global_step": 912, "epoch": 0.03521779425393883, "loss": -0.0072, "grad_norm": 2.130798816680908, "learning_rate": 7.2393939393939404e-06, "num_tokens": 1960780.0, "completions/mean_length": 51.375, "completions/min_length": 50.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.991952121257782, "rewards/meter/std": 0.008988977409899235, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.991952121257782, "rewards/total_composite/std": 0.008988977409899235, "reward": 0.991952121257782, "reward_std": 0.008988974615931511, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0225484948605299, "sampling/sampling_logp_difference/max": 1.0655335187911987, "sampling/importance_sampling_ratio/min": 0.3445439636707306, "sampling/importance_sampling_ratio/mean": 0.9964022040367126, "sampling/importance_sampling_ratio/max": 1.5943405628204346, "entropy": 0.09524842072278261, "clip_ratio/low_mean": 0.0024999999441206455, "clip_ratio/low_min": 0.0024999999441206455, "clip_ratio/high_mean": 0.021779576083645225, "clip_ratio/high_max": 0.021779576083645225, "clip_ratio/region_mean": 0.02427957602776587, "reward_total_mean": 0.991952121257782, "reward_meter_mean": 0.991952121257782, "reward_meter_std": 0.008988977409899235, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.991952121257782, "reward_total_composite_std": 0.008988977409899235, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 912.0} {"timestamp_utc": "2026-04-11T21:21:43Z", "mode": "train", "global_step": 913, "epoch": 0.035256410256410256, "loss": 0.0033, "grad_norm": 5.4082231521606445, "learning_rate": 7.236363636363637e-06, "num_tokens": 1962873.0, "completions/mean_length": 92.625, "completions/min_length": 92.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.625, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.8067741394042969, "rewards/meter/std": 0.175697460770607, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8067741394042969, "rewards/total_composite/std": 0.175697460770607, "reward": 0.8067741394042969, "reward_std": 0.1756974458694458, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01361581776291132, "sampling/sampling_logp_difference/max": 1.826080560684204, "sampling/importance_sampling_ratio/min": 0.16104353964328766, "sampling/importance_sampling_ratio/mean": 1.0001976490020752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.026704848161898553, "clip_ratio/low_mean": 0.004046867717988789, "clip_ratio/low_min": 0.004046867717988789, "clip_ratio/high_mean": 0.010811748332343996, "clip_ratio/high_max": 0.010811748332343996, "clip_ratio/region_mean": 0.014858616050332785, "reward_total_mean": 0.8067741394042969, "reward_meter_mean": 0.8067741394042969, "reward_meter_std": 0.175697460770607, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8067741394042969, "reward_total_composite_std": 0.175697460770607, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 913.0} {"timestamp_utc": "2026-04-11T21:21:53Z", "mode": "train", "global_step": 914, "epoch": 0.03529502625888168, "loss": -0.2899, "grad_norm": 0.30297282338142395, "learning_rate": 7.233333333333334e-06, "num_tokens": 1966854.0, "completions/mean_length": 355.625, "completions/min_length": 322.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 333.2857360839844, "completions/min_terminated_length": 322.0, "completions/max_terminated_length": 337.0, "rewards/meter/mean": 0.9955409169197083, "rewards/meter/std": 0.0037022808101028204, "rewards/count_adherence/mean": 0.7708333134651184, "rewards/count_adherence/std": 0.1767766773700714, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.7268685698509216, "rewards/total_composite/std": 0.2936992943286896, "reward": 0.7268685698509216, "reward_std": 0.2936992943286896, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002380110090598464, "sampling/sampling_logp_difference/max": 1.0458037853240967, "sampling/importance_sampling_ratio/min": 0.35140925645828247, "sampling/importance_sampling_ratio/mean": 1.0005160570144653, "sampling/importance_sampling_ratio/max": 1.2111318111419678, "entropy": 0.015034663956612349, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0007668711477890611, "clip_ratio/high_max": 0.0007668711477890611, "clip_ratio/region_mean": 0.0007668711477890611, "reward_total_mean": 0.7268685698509216, "reward_meter_mean": 0.9955409169197083, "reward_meter_std": 0.0037022808101028204, "reward_count_adherence_mean": 0.7708333134651184, "reward_count_adherence_std": 0.1767766773700714, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.7268685698509216, "reward_total_composite_std": 0.2936992943286896, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 914.0} {"timestamp_utc": "2026-04-11T21:21:58Z", "mode": "train", "global_step": 915, "epoch": 0.035333642261353104, "loss": -0.015, "grad_norm": 11.037290573120117, "learning_rate": 7.2303030303030305e-06, "num_tokens": 1968333.0, "completions/mean_length": 31.875, "completions/min_length": 28.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.5682185292243958, "rewards/meter/std": 0.15682178735733032, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5682185292243958, "rewards/total_composite/std": 0.15682178735733032, "reward": 0.5682185292243958, "reward_std": 0.15682180225849152, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04318798705935478, "sampling/sampling_logp_difference/max": 1.5100479125976562, "sampling/importance_sampling_ratio/min": 0.22089938819408417, "sampling/importance_sampling_ratio/mean": 1.0079559087753296, "sampling/importance_sampling_ratio/max": 1.9600094556808472, "entropy": 0.1962036518380046, "clip_ratio/low_mean": 0.02046730974689126, "clip_ratio/low_min": 0.02046730974689126, "clip_ratio/high_mean": 0.01905776560306549, "clip_ratio/high_max": 0.01905776560306549, "clip_ratio/region_mean": 0.03952507534995675, "reward_total_mean": 0.5682185292243958, "reward_meter_mean": 0.5682185292243958, "reward_meter_std": 0.15682178735733032, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5682185292243958, "reward_total_composite_std": 0.15682178735733032, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 915.0} {"timestamp_utc": "2026-04-11T21:22:05Z", "mode": "train", "global_step": 916, "epoch": 0.03537225826382453, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.227272727272729e-06, "num_tokens": 1971493.0, "completions/mean_length": 211.0, "completions/min_length": 211.0, "completions/max_length": 211.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 211.0, "completions/min_terminated_length": 211.0, "completions/max_terminated_length": 211.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002760888892225921, "sampling/sampling_logp_difference/max": 0.11726605892181396, "sampling/importance_sampling_ratio/min": 0.889348566532135, "sampling/importance_sampling_ratio/mean": 1.0000563859939575, "sampling/importance_sampling_ratio/max": 1.0166982412338257, "entropy": 0.001997560597374104, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 916.0} {"timestamp_utc": "2026-04-11T21:22:09Z", "mode": "train", "global_step": 917, "epoch": 0.03541087426629595, "loss": 0.002, "grad_norm": 15.208098411560059, "learning_rate": 7.224242424242425e-06, "num_tokens": 1972900.0, "completions/mean_length": 33.875, "completions/min_length": 31.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.875, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9886395931243896, "rewards/meter/std": 0.005185124464333057, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9886395931243896, "rewards/total_composite/std": 0.005185124464333057, "reward": 0.9886395931243896, "reward_std": 0.005185126792639494, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.045671071857213974, "sampling/sampling_logp_difference/max": 3.7692480087280273, "sampling/importance_sampling_ratio/min": 0.023069405928254128, "sampling/importance_sampling_ratio/mean": 0.9981818199157715, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15464295260608196, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.01895968639291823, "clip_ratio/high_max": 0.01895968639291823, "clip_ratio/region_mean": 0.01895968639291823, "reward_total_mean": 0.9886395931243896, "reward_meter_mean": 0.9886395931243896, "reward_meter_std": 0.005185124464333057, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9886395931243896, "reward_total_composite_std": 0.005185124464333057, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 917.0} {"timestamp_utc": "2026-04-11T21:22:14Z", "mode": "train", "global_step": 918, "epoch": 0.03544949026876738, "loss": -0.0068, "grad_norm": 3.0432510375976562, "learning_rate": 7.221212121212122e-06, "num_tokens": 1974839.0, "completions/mean_length": 78.375, "completions/min_length": 76.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.375, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9951028823852539, "rewards/meter/std": 0.0006839877460151911, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9951028823852539, "rewards/total_composite/std": 0.0006839877460151911, "reward": 0.9951028823852539, "reward_std": 0.0006839786074124277, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01417941227555275, "sampling/sampling_logp_difference/max": 0.5501952171325684, "sampling/importance_sampling_ratio/min": 0.5961045026779175, "sampling/importance_sampling_ratio/mean": 1.0006179809570312, "sampling/importance_sampling_ratio/max": 1.7335914373397827, "entropy": 0.08995178993791342, "clip_ratio/low_mean": 0.0048326835967600346, "clip_ratio/low_min": 0.0048326835967600346, "clip_ratio/high_mean": 0.006395183620043099, "clip_ratio/high_max": 0.006395183620043099, "clip_ratio/region_mean": 0.011227867216803133, "reward_total_mean": 0.9951028823852539, "reward_meter_mean": 0.9951028823852539, "reward_meter_std": 0.0006839877460151911, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9951028823852539, "reward_total_composite_std": 0.0006839877460151911, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 918.0} {"timestamp_utc": "2026-04-11T21:22:21Z", "mode": "train", "global_step": 919, "epoch": 0.0354881062712388, "loss": 0.0795, "grad_norm": 4.4958624839782715, "learning_rate": 7.218181818181819e-06, "num_tokens": 1976737.0, "completions/mean_length": 73.25, "completions/min_length": 69.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.25, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.5712785124778748, "rewards/meter/std": 0.47130051255226135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5712785124778748, "rewards/total_composite/std": 0.47130051255226135, "reward": 0.5712785124778748, "reward_std": 0.47130051255226135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.043649815022945404, "sampling/sampling_logp_difference/max": 6.4018964767456055, "sampling/importance_sampling_ratio/min": 0.001658409251831472, "sampling/importance_sampling_ratio/mean": 0.9948803186416626, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06092626950703561, "clip_ratio/low_mean": 0.007797118509188294, "clip_ratio/low_min": 0.007797118509188294, "clip_ratio/high_mean": 0.009057971183210611, "clip_ratio/high_max": 0.009057971183210611, "clip_ratio/region_mean": 0.016855089692398906, "reward_total_mean": 0.5712785124778748, "reward_meter_mean": 0.5712785124778748, "reward_meter_std": 0.47130051255226135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5712785124778748, "reward_total_composite_std": 0.47130051255226135, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 919.0} {"timestamp_utc": "2026-04-11T21:22:25Z", "mode": "train", "global_step": 920, "epoch": 0.035526722273710225, "loss": -0.0032, "grad_norm": 3.2345025539398193, "learning_rate": 7.215151515151516e-06, "num_tokens": 1978500.0, "completions/mean_length": 65.375, "completions/min_length": 65.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9969853162765503, "rewards/meter/std": 0.0002019107632804662, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9969853162765503, "rewards/total_composite/std": 0.0002019107632804662, "reward": 0.9969853162765503, "reward_std": 0.00020190946816001087, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007577007170766592, "sampling/sampling_logp_difference/max": 1.625145435333252, "sampling/importance_sampling_ratio/min": 0.19688303768634796, "sampling/importance_sampling_ratio/mean": 1.0003613233566284, "sampling/importance_sampling_ratio/max": 1.7756403684616089, "entropy": 0.027456025825813413, "clip_ratio/low_mean": 0.005769230774603784, "clip_ratio/low_min": 0.005769230774603784, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/region_mean": 0.00766317022498697, "reward_total_mean": 0.9969853162765503, "reward_meter_mean": 0.9969853162765503, "reward_meter_std": 0.0002019107632804662, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9969853162765503, "reward_total_composite_std": 0.0002019107632804662, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 920.0} {"timestamp_utc": "2026-04-11T21:22:30Z", "mode": "train", "global_step": 921, "epoch": 0.03556533827618165, "loss": 0.0565, "grad_norm": 11.585659980773926, "learning_rate": 7.212121212121212e-06, "num_tokens": 1980162.0, "completions/mean_length": 60.75, "completions/min_length": 56.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8528519868850708, "rewards/meter/std": 0.3407958447933197, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8528519868850708, "rewards/total_composite/std": 0.3407958447933197, "reward": 0.8528519868850708, "reward_std": 0.3407958447933197, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036973193287849426, "sampling/sampling_logp_difference/max": 2.069132089614868, "sampling/importance_sampling_ratio/min": 0.12629535794258118, "sampling/importance_sampling_ratio/mean": 1.0029891729354858, "sampling/importance_sampling_ratio/max": 1.7221803665161133, "entropy": 0.229447185061872, "clip_ratio/low_mean": 0.0036231884732842445, "clip_ratio/low_min": 0.0036231884732842445, "clip_ratio/high_mean": 0.012532679131254554, "clip_ratio/high_max": 0.012532679131254554, "clip_ratio/region_mean": 0.0161558676045388, "reward_total_mean": 0.8528519868850708, "reward_meter_mean": 0.8528519868850708, "reward_meter_std": 0.3407958447933197, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8528519868850708, "reward_total_composite_std": 0.3407958447933197, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 921.0} {"timestamp_utc": "2026-04-11T21:22:36Z", "mode": "train", "global_step": 922, "epoch": 0.03560395427865307, "loss": 0.0011, "grad_norm": 0.6000277400016785, "learning_rate": 7.2090909090909104e-06, "num_tokens": 1982842.0, "completions/mean_length": 160.0, "completions/min_length": 160.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.0, "completions/min_terminated_length": 160.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9639455080032349, "rewards/meter/std": 0.001239220960997045, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8434523344039917, "rewards/total_composite/std": 0.001084321876987815, "reward": 0.8434523344039917, "reward_std": 0.0010843126801773906, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002822460373863578, "sampling/sampling_logp_difference/max": 1.7144312858581543, "sampling/importance_sampling_ratio/min": 0.18006610870361328, "sampling/importance_sampling_ratio/mean": 0.9997346997261047, "sampling/importance_sampling_ratio/max": 1.1099183559417725, "entropy": 0.012898905668407679, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0015625000232830644, "clip_ratio/high_max": 0.0015625000232830644, "clip_ratio/region_mean": 0.0015625000232830644, "reward_total_mean": 0.8434523344039917, "reward_meter_mean": 0.9639455080032349, "reward_meter_std": 0.001239220960997045, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8434523344039917, "reward_total_composite_std": 0.001084321876987815, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 922.0} {"timestamp_utc": "2026-04-11T21:22:42Z", "mode": "train", "global_step": 923, "epoch": 0.0356425702811245, "loss": 0.0145, "grad_norm": 5.261467933654785, "learning_rate": 7.206060606060606e-06, "num_tokens": 1984684.0, "completions/mean_length": 66.25, "completions/min_length": 65.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9920624494552612, "rewards/meter/std": 0.005796308629214764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9920624494552612, "rewards/total_composite/std": 0.005796308629214764, "reward": 0.9920624494552612, "reward_std": 0.005796318408101797, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015619166195392609, "sampling/sampling_logp_difference/max": 1.09493887424469, "sampling/importance_sampling_ratio/min": 0.3345600664615631, "sampling/importance_sampling_ratio/mean": 1.001530408859253, "sampling/importance_sampling_ratio/max": 1.528792142868042, "entropy": 0.07848310004919767, "clip_ratio/low_mean": 0.0018115942366421223, "clip_ratio/low_min": 0.0018115942366421223, "clip_ratio/high_mean": 0.007692307699471712, "clip_ratio/high_max": 0.007692307699471712, "clip_ratio/region_mean": 0.009503901936113834, "reward_total_mean": 0.9920624494552612, "reward_meter_mean": 0.9920624494552612, "reward_meter_std": 0.005796308629214764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9920624494552612, "reward_total_composite_std": 0.005796308629214764, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 923.0} {"timestamp_utc": "2026-04-11T21:22:46Z", "mode": "train", "global_step": 924, "epoch": 0.03568118628359592, "loss": -0.0486, "grad_norm": 9.44915771484375, "learning_rate": 7.203030303030304e-06, "num_tokens": 1986267.0, "completions/mean_length": 36.875, "completions/min_length": 36.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.3160516023635864, "rewards/meter/std": 0.26414522528648376, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3160516023635864, "rewards/total_composite/std": 0.26414522528648376, "reward": 0.3160516023635864, "reward_std": 0.26414522528648376, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021508069708943367, "sampling/sampling_logp_difference/max": 1.3987207412719727, "sampling/importance_sampling_ratio/min": 0.2469126284122467, "sampling/importance_sampling_ratio/mean": 0.990954577922821, "sampling/importance_sampling_ratio/max": 1.295661449432373, "entropy": 0.02673885691910982, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/high_mean": 0.0029069767333567142, "clip_ratio/high_max": 0.0029069767333567142, "clip_ratio/region_mean": 0.006379198981449008, "reward_total_mean": 0.3160516023635864, "reward_meter_mean": 0.3160516023635864, "reward_meter_std": 0.26414522528648376, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3160516023635864, "reward_total_composite_std": 0.26414522528648376, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 924.0} {"timestamp_utc": "2026-04-11T21:22:51Z", "mode": "train", "global_step": 925, "epoch": 0.035719802286067345, "loss": -0.0017, "grad_norm": 5.008955955505371, "learning_rate": 7.2000000000000005e-06, "num_tokens": 1987971.0, "completions/mean_length": 55.0, "completions/min_length": 54.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.975671648979187, "rewards/meter/std": 0.006744861137121916, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.975671648979187, "rewards/total_composite/std": 0.006744861137121916, "reward": 0.975671648979187, "reward_std": 0.00674486206844449, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018869103863835335, "sampling/sampling_logp_difference/max": 1.7920613288879395, "sampling/importance_sampling_ratio/min": 0.16661636531352997, "sampling/importance_sampling_ratio/mean": 0.999152421951294, "sampling/importance_sampling_ratio/max": 1.604773998260498, "entropy": 0.07793995039537549, "clip_ratio/low_mean": 0.00909090880304575, "clip_ratio/low_min": 0.00909090880304575, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.00909090880304575, "reward_total_mean": 0.975671648979187, "reward_meter_mean": 0.975671648979187, "reward_meter_std": 0.006744861137121916, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.975671648979187, "reward_total_composite_std": 0.006744861137121916, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 925.0} {"timestamp_utc": "2026-04-11T21:23:01Z", "mode": "train", "global_step": 926, "epoch": 0.03575841828853877, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.196969696969698e-06, "num_tokens": 1989755.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9969740509986877, "rewards/meter/std": 1.276752391277114e-05, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7975792288780212, "rewards/total_composite/std": 1.0207570994680282e-05, "reward": 0.7975792288780212, "reward_std": 1.0214442227152176e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7975792288780212, "reward_meter_mean": 0.9969740509986877, "reward_meter_std": 1.276752391277114e-05, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7975792288780212, "reward_total_composite_std": 1.0207570994680282e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 926.0} {"timestamp_utc": "2026-04-11T21:23:11Z", "mode": "train", "global_step": 927, "epoch": 0.035797034291010194, "loss": -0.0789, "grad_norm": 0.5503115057945251, "learning_rate": 7.193939393939394e-06, "num_tokens": 1991062.0, "completions/mean_length": 150.375, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 29.83333396911621, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.7567366361618042, "rewards/meter/std": 0.4422414302825928, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.7464744448661804, "rewards/total_composite/std": 0.4607347249984741, "reward": 0.7464744448661804, "reward_std": 0.4607347249984741, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03661965951323509, "sampling/sampling_logp_difference/max": 1.0455524921417236, "sampling/importance_sampling_ratio/min": 0.3514975607395172, "sampling/importance_sampling_ratio/mean": 0.9947470426559448, "sampling/importance_sampling_ratio/max": 1.4951473474502563, "entropy": 0.08896318543702364, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.020833334419876337, "clip_ratio/high_max": 0.020833334419876337, "clip_ratio/region_mean": 0.020833334419876337, "reward_total_mean": 0.7464744448661804, "reward_meter_mean": 0.7567366361618042, "reward_meter_std": 0.4422414302825928, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.7464744448661804, "reward_total_composite_std": 0.4607347249984741, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 927.0} {"timestamp_utc": "2026-04-11T21:23:16Z", "mode": "train", "global_step": 928, "epoch": 0.03583565029348162, "loss": 0.0034, "grad_norm": 3.169527292251587, "learning_rate": 7.1909090909090914e-06, "num_tokens": 1992687.0, "completions/mean_length": 61.125, "completions/min_length": 61.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9981940984725952, "rewards/meter/std": 0.0011257634032517672, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9981940984725952, "rewards/total_composite/std": 0.0011257634032517672, "reward": 0.9981940984725952, "reward_std": 0.0011257664300501347, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003506365930661559, "sampling/sampling_logp_difference/max": 0.3527231216430664, "sampling/importance_sampling_ratio/min": 0.7537627220153809, "sampling/importance_sampling_ratio/mean": 1.0007489919662476, "sampling/importance_sampling_ratio/max": 1.4229371547698975, "entropy": 0.01905843592248857, "clip_ratio/low_mean": 0.006048386916518211, "clip_ratio/low_min": 0.006048386916518211, "clip_ratio/high_mean": 0.006147540640085936, "clip_ratio/high_max": 0.006147540640085936, "clip_ratio/region_mean": 0.012195927556604147, "reward_total_mean": 0.9981940984725952, "reward_meter_mean": 0.9981940984725952, "reward_meter_std": 0.0011257634032517672, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9981940984725952, "reward_total_composite_std": 0.0011257634032517672, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 928.0} {"timestamp_utc": "2026-04-11T21:23:21Z", "mode": "train", "global_step": 929, "epoch": 0.03587426629595304, "loss": 0.0455, "grad_norm": 6.427262306213379, "learning_rate": 7.187878787878788e-06, "num_tokens": 1994597.0, "completions/mean_length": 77.75, "completions/min_length": 71.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.75, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.013365390710532665, "rewards/meter/std": 0.024956075474619865, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.013362133875489235, "rewards/total_composite/std": 0.024958059191703796, "reward": 0.013362133875489235, "reward_std": 0.024958059191703796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.057392947375774384, "sampling/sampling_logp_difference/max": 4.0821919441223145, "sampling/importance_sampling_ratio/min": 0.016870446503162384, "sampling/importance_sampling_ratio/mean": 0.9932682514190674, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11699613835662603, "clip_ratio/low_mean": 0.01578302902635187, "clip_ratio/low_min": 0.01578302902635187, "clip_ratio/high_mean": 0.006767879938706756, "clip_ratio/high_max": 0.006767879938706756, "clip_ratio/region_mean": 0.022550908965058625, "reward_total_mean": 0.013362133875489235, "reward_meter_mean": 0.013365390710532665, "reward_meter_std": 0.024956075474619865, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.013362133875489235, "reward_total_composite_std": 0.024958059191703796, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 929.0} {"timestamp_utc": "2026-04-11T21:23:26Z", "mode": "train", "global_step": 930, "epoch": 0.035912882298424466, "loss": 0.0208, "grad_norm": 13.685571670532227, "learning_rate": 7.184848484848486e-06, "num_tokens": 1996382.0, "completions/mean_length": 65.125, "completions/min_length": 64.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7817329168319702, "rewards/meter/std": 0.3914225399494171, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7817329168319702, "rewards/total_composite/std": 0.3914225399494171, "reward": 0.7817329168319702, "reward_std": 0.3914225399494171, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030531782656908035, "sampling/sampling_logp_difference/max": 4.1985554695129395, "sampling/importance_sampling_ratio/min": 0.015017255209386349, "sampling/importance_sampling_ratio/mean": 0.9985852837562561, "sampling/importance_sampling_ratio/max": 1.6812983751296997, "entropy": 0.05809917626902461, "clip_ratio/low_mean": 0.005711825448088348, "clip_ratio/low_min": 0.005711825448088348, "clip_ratio/high_mean": 0.007722355774603784, "clip_ratio/high_max": 0.007722355774603784, "clip_ratio/region_mean": 0.013434181222692132, "reward_total_mean": 0.7817329168319702, "reward_meter_mean": 0.7817329168319702, "reward_meter_std": 0.3914225399494171, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7817329168319702, "reward_total_composite_std": 0.3914225399494171, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 930.0} {"timestamp_utc": "2026-04-11T21:23:32Z", "mode": "train", "global_step": 931, "epoch": 0.03595149830089589, "loss": -0.026, "grad_norm": 1.78580904006958, "learning_rate": 7.181818181818182e-06, "num_tokens": 1999543.0, "completions/mean_length": 179.125, "completions/min_length": 166.0, "completions/max_length": 181.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 179.125, "completions/min_terminated_length": 166.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.9984796047210693, "rewards/meter/std": 0.00031818763818591833, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9776943325996399, "rewards/total_composite/std": 0.05910775437951088, "reward": 0.9776943325996399, "reward_std": 0.05910775065422058, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003947978839278221, "sampling/sampling_logp_difference/max": 3.3655812740325928, "sampling/importance_sampling_ratio/min": 0.03454193100333214, "sampling/importance_sampling_ratio/mean": 0.9992736577987671, "sampling/importance_sampling_ratio/max": 1.1424589157104492, "entropy": 0.007839636178687215, "clip_ratio/low_mean": 0.0007530120201408863, "clip_ratio/low_min": 0.0007530120201408863, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0007530120201408863, "reward_total_mean": 0.9776943325996399, "reward_meter_mean": 0.9984796047210693, "reward_meter_std": 0.00031818763818591833, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9776943325996399, "reward_total_composite_std": 0.05910775437951088, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 931.0} {"timestamp_utc": "2026-04-11T21:23:37Z", "mode": "train", "global_step": 932, "epoch": 0.035990114303367314, "loss": 0.0229, "grad_norm": 8.503756523132324, "learning_rate": 7.17878787878788e-06, "num_tokens": 2001129.0, "completions/mean_length": 55.25, "completions/min_length": 51.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.6907171010971069, "rewards/meter/std": 0.3268168568611145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6907171010971069, "rewards/total_composite/std": 0.3268168568611145, "reward": 0.6907171010971069, "reward_std": 0.3268168568611145, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06431894749403, "sampling/sampling_logp_difference/max": 2.275351047515869, "sampling/importance_sampling_ratio/min": 0.10276082903146744, "sampling/importance_sampling_ratio/mean": 0.9887388944625854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22322631068527699, "clip_ratio/low_mean": 0.026154047809541225, "clip_ratio/low_min": 0.026154047809541225, "clip_ratio/high_mean": 0.05068119755014777, "clip_ratio/high_max": 0.05068119755014777, "clip_ratio/region_mean": 0.076835245359689, "reward_total_mean": 0.6907171010971069, "reward_meter_mean": 0.6907171010971069, "reward_meter_std": 0.3268168568611145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6907171010971069, "reward_total_composite_std": 0.3268168568611145, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 932.0} {"timestamp_utc": "2026-04-11T21:23:42Z", "mode": "train", "global_step": 933, "epoch": 0.03602873030583874, "loss": 0.0576, "grad_norm": 10.403377532958984, "learning_rate": 7.175757575757576e-06, "num_tokens": 2002926.0, "completions/mean_length": 55.625, "completions/min_length": 52.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.625, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7252499461174011, "rewards/meter/std": 0.3348124623298645, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7252499461174011, "rewards/total_composite/std": 0.3348124623298645, "reward": 0.7252499461174011, "reward_std": 0.3348124325275421, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07079855352640152, "sampling/sampling_logp_difference/max": 4.724750518798828, "sampling/importance_sampling_ratio/min": 0.00887292716652155, "sampling/importance_sampling_ratio/mean": 0.9955315589904785, "sampling/importance_sampling_ratio/max": 1.9811956882476807, "entropy": 0.11601806059479713, "clip_ratio/low_mean": 0.008131720358505845, "clip_ratio/low_min": 0.008131720358505845, "clip_ratio/high_mean": 0.025539263151586056, "clip_ratio/high_max": 0.025539263151586056, "clip_ratio/region_mean": 0.0336709835100919, "reward_total_mean": 0.7252499461174011, "reward_meter_mean": 0.7252499461174011, "reward_meter_std": 0.3348124623298645, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7252499461174011, "reward_total_composite_std": 0.3348124623298645, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 933.0} {"timestamp_utc": "2026-04-11T21:23:46Z", "mode": "train", "global_step": 934, "epoch": 0.03606734630831016, "loss": 0.0111, "grad_norm": 10.28354549407959, "learning_rate": 7.172727272727273e-06, "num_tokens": 2004665.0, "completions/mean_length": 60.375, "completions/min_length": 58.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.375, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.04125343635678291, "rewards/meter/std": 0.0725683942437172, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.04125343635678291, "rewards/total_composite/std": 0.0725683942437172, "reward": 0.04125343635678291, "reward_std": 0.0725683942437172, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06495323032140732, "sampling/sampling_logp_difference/max": 5.703890800476074, "sampling/importance_sampling_ratio/min": 0.0033329720608890057, "sampling/importance_sampling_ratio/mean": 1.0052381753921509, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1609570113942027, "clip_ratio/low_mean": 0.018655708292499185, "clip_ratio/low_min": 0.018655708292499185, "clip_ratio/high_mean": 0.004166666883975267, "clip_ratio/high_max": 0.004166666883975267, "clip_ratio/region_mean": 0.022822375176474452, "reward_total_mean": 0.04125343635678291, "reward_meter_mean": 0.04125343635678291, "reward_meter_std": 0.0725683942437172, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.04125343635678291, "reward_total_composite_std": 0.0725683942437172, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 934.0} {"timestamp_utc": "2026-04-11T21:23:51Z", "mode": "train", "global_step": 935, "epoch": 0.036105962310781586, "loss": 0.0344, "grad_norm": 18.004487991333008, "learning_rate": 7.16969696969697e-06, "num_tokens": 2006015.0, "completions/mean_length": 32.75, "completions/min_length": 30.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.75, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.28016990423202515, "rewards/meter/std": 0.442457914352417, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.28016990423202515, "rewards/total_composite/std": 0.442457914352417, "reward": 0.28016990423202515, "reward_std": 0.4424578845500946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02748202532529831, "sampling/sampling_logp_difference/max": 1.1870431900024414, "sampling/importance_sampling_ratio/min": 0.5575820207595825, "sampling/importance_sampling_ratio/mean": 1.0117889642715454, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10954149113968015, "clip_ratio/low_mean": 0.02239941479638219, "clip_ratio/low_min": 0.02239941479638219, "clip_ratio/high_mean": 0.008333333767950535, "clip_ratio/high_max": 0.008333333767950535, "clip_ratio/region_mean": 0.030732748564332724, "reward_total_mean": 0.28016990423202515, "reward_meter_mean": 0.28016990423202515, "reward_meter_std": 0.442457914352417, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.28016990423202515, "reward_total_composite_std": 0.442457914352417, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 935.0} {"timestamp_utc": "2026-04-11T21:23:57Z", "mode": "train", "global_step": 936, "epoch": 0.03614457831325301, "loss": 0.0221, "grad_norm": 6.936933994293213, "learning_rate": 7.166666666666667e-06, "num_tokens": 2008376.0, "completions/mean_length": 102.125, "completions/min_length": 96.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.125, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.8500412702560425, "rewards/meter/std": 0.32248687744140625, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8500412702560425, "rewards/total_composite/std": 0.32248687744140625, "reward": 0.8500412702560425, "reward_std": 0.32248687744140625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011623694561421871, "sampling/sampling_logp_difference/max": 0.9906949996948242, "sampling/importance_sampling_ratio/min": 0.3713185489177704, "sampling/importance_sampling_ratio/mean": 0.9981279373168945, "sampling/importance_sampling_ratio/max": 1.3035365343093872, "entropy": 0.06633846508339047, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/high_mean": 0.0050833027344197035, "clip_ratio/high_max": 0.0050833027344197035, "clip_ratio/region_mean": 0.007398117566481233, "reward_total_mean": 0.8500412702560425, "reward_meter_mean": 0.8500412702560425, "reward_meter_std": 0.32248687744140625, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8500412702560425, "reward_total_composite_std": 0.32248687744140625, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 936.0} {"timestamp_utc": "2026-04-11T21:24:02Z", "mode": "train", "global_step": 937, "epoch": 0.036183194315724435, "loss": -0.0002, "grad_norm": 1.914229154586792, "learning_rate": 7.163636363636363e-06, "num_tokens": 2010312.0, "completions/mean_length": 81.0, "completions/min_length": 81.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.0, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9976653456687927, "rewards/meter/std": 0.0004223364812787622, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6651102304458618, "rewards/total_composite/std": 0.0002815788902807981, "reward": 0.6651102304458618, "reward_std": 0.0002815948100760579, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0029760051984339952, "sampling/sampling_logp_difference/max": 0.4011359214782715, "sampling/importance_sampling_ratio/min": 0.6695590019226074, "sampling/importance_sampling_ratio/mean": 0.9999629259109497, "sampling/importance_sampling_ratio/max": 1.1242009401321411, "entropy": 0.017346444772556424, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0015432098880410194, "clip_ratio/high_max": 0.0015432098880410194, "clip_ratio/region_mean": 0.0015432098880410194, "reward_total_mean": 0.6651102304458618, "reward_meter_mean": 0.9976653456687927, "reward_meter_std": 0.0004223364812787622, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6651102304458618, "reward_total_composite_std": 0.0002815788902807981, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 937.0} {"timestamp_utc": "2026-04-11T21:24:08Z", "mode": "train", "global_step": 938, "epoch": 0.03622181031819586, "loss": -0.0404, "grad_norm": 1.588853120803833, "learning_rate": 7.1606060606060615e-06, "num_tokens": 2013000.0, "completions/mean_length": 157.0, "completions/min_length": 145.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.0, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.997098445892334, "rewards/meter/std": 0.0015495484694838524, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9473224878311157, "rewards/total_composite/std": 0.09314737468957901, "reward": 0.9473224878311157, "reward_std": 0.0931473970413208, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0039437501691281796, "sampling/sampling_logp_difference/max": 0.5851579904556274, "sampling/importance_sampling_ratio/min": 0.5570178627967834, "sampling/importance_sampling_ratio/mean": 0.9993342757225037, "sampling/importance_sampling_ratio/max": 1.2282726764678955, "entropy": 0.016126448521390557, "clip_ratio/low_mean": 0.0017241379246115685, "clip_ratio/low_min": 0.0017241379246115685, "clip_ratio/high_mean": 0.002329192589968443, "clip_ratio/high_max": 0.002329192589968443, "clip_ratio/region_mean": 0.004053330514580011, "reward_total_mean": 0.9473224878311157, "reward_meter_mean": 0.997098445892334, "reward_meter_std": 0.0015495484694838524, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9473224878311157, "reward_total_composite_std": 0.09314737468957901, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 938.0} {"timestamp_utc": "2026-04-11T21:24:13Z", "mode": "train", "global_step": 939, "epoch": 0.03626042632066728, "loss": 0.0551, "grad_norm": 15.364835739135742, "learning_rate": 7.157575757575758e-06, "num_tokens": 2014470.0, "completions/mean_length": 27.75, "completions/min_length": 26.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.6426187753677368, "rewards/meter/std": 0.3321544826030731, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6426187753677368, "rewards/total_composite/std": 0.3321544826030731, "reward": 0.6426187753677368, "reward_std": 0.33215445280075073, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07284780591726303, "sampling/sampling_logp_difference/max": 1.4266352653503418, "sampling/importance_sampling_ratio/min": 0.24011549353599548, "sampling/importance_sampling_ratio/mean": 1.0016535520553589, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2934161014854908, "clip_ratio/low_mean": 0.026190476957708597, "clip_ratio/low_min": 0.026190476957708597, "clip_ratio/high_mean": 0.040990028996020555, "clip_ratio/high_max": 0.040990028996020555, "clip_ratio/region_mean": 0.06718050595372915, "reward_total_mean": 0.6426187753677368, "reward_meter_mean": 0.6426187753677368, "reward_meter_std": 0.3321544826030731, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6426187753677368, "reward_total_composite_std": 0.3321544826030731, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 939.0} {"timestamp_utc": "2026-04-11T21:24:18Z", "mode": "train", "global_step": 940, "epoch": 0.03629904232313871, "loss": -0.012, "grad_norm": 2.963726043701172, "learning_rate": 7.154545454545455e-06, "num_tokens": 2016275.0, "completions/mean_length": 72.625, "completions/min_length": 72.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.625, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.941265881061554, "rewards/meter/std": 0.013865120708942413, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.941265881061554, "rewards/total_composite/std": 0.013865120708942413, "reward": 0.941265881061554, "reward_std": 0.013865115121006966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007979449816048145, "sampling/sampling_logp_difference/max": 1.5069544315338135, "sampling/importance_sampling_ratio/min": 0.2215837985277176, "sampling/importance_sampling_ratio/mean": 1.0008445978164673, "sampling/importance_sampling_ratio/max": 1.3666635751724243, "entropy": 0.028274629497900605, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/high_mean": 0.003359487745910883, "clip_ratio/high_max": 0.003359487745910883, "clip_ratio/region_mean": 0.00509559886995703, "reward_total_mean": 0.941265881061554, "reward_meter_mean": 0.941265881061554, "reward_meter_std": 0.013865120708942413, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.941265881061554, "reward_total_composite_std": 0.013865120708942413, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 940.0} {"timestamp_utc": "2026-04-11T21:24:22Z", "mode": "train", "global_step": 941, "epoch": 0.03633765832561013, "loss": -0.0139, "grad_norm": 11.151459693908691, "learning_rate": 7.151515151515152e-06, "num_tokens": 2018046.0, "completions/mean_length": 41.375, "completions/min_length": 37.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.375, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.7161558866500854, "rewards/meter/std": 0.3421039283275604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7161558866500854, "rewards/total_composite/std": 0.3421039283275604, "reward": 0.7161558866500854, "reward_std": 0.3421039283275604, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0714624896645546, "sampling/sampling_logp_difference/max": 1.5439640283584595, "sampling/importance_sampling_ratio/min": 0.21353298425674438, "sampling/importance_sampling_ratio/mean": 0.9894468188285828, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23116246238350868, "clip_ratio/low_mean": 0.016638513887301087, "clip_ratio/low_min": 0.016638513887301087, "clip_ratio/high_mean": 0.0419863504357636, "clip_ratio/high_max": 0.0419863504357636, "clip_ratio/region_mean": 0.058624864323064685, "reward_total_mean": 0.7161558866500854, "reward_meter_mean": 0.7161558866500854, "reward_meter_std": 0.3421039283275604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7161558866500854, "reward_total_composite_std": 0.3421039283275604, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 941.0} {"timestamp_utc": "2026-04-11T21:24:28Z", "mode": "train", "global_step": 942, "epoch": 0.036376274328081555, "loss": 0.003, "grad_norm": 0.48789122700691223, "learning_rate": 7.148484848484849e-06, "num_tokens": 2020497.0, "completions/mean_length": 128.375, "completions/min_length": 123.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.375, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9937758445739746, "rewards/meter/std": 0.012314152903854847, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9937758445739746, "rewards/total_composite/std": 0.012314152903854847, "reward": 0.9937758445739746, "reward_std": 0.012314137071371078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003727847710251808, "sampling/sampling_logp_difference/max": 0.7433128356933594, "sampling/importance_sampling_ratio/min": 0.47553592920303345, "sampling/importance_sampling_ratio/mean": 1.0006505250930786, "sampling/importance_sampling_ratio/max": 1.4588483572006226, "entropy": 0.017605803557671607, "clip_ratio/low_mean": 0.000961538462433964, "clip_ratio/low_min": 0.000961538462433964, "clip_ratio/high_mean": 0.0019379844889044762, "clip_ratio/high_max": 0.0019379844889044762, "clip_ratio/region_mean": 0.00289952295133844, "reward_total_mean": 0.9937758445739746, "reward_meter_mean": 0.9937758445739746, "reward_meter_std": 0.012314152903854847, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9937758445739746, "reward_total_composite_std": 0.012314152903854847, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 942.0} {"timestamp_utc": "2026-04-11T21:24:36Z", "mode": "train", "global_step": 943, "epoch": 0.03641489033055298, "loss": -0.0001, "grad_norm": 0.140433207154274, "learning_rate": 7.145454545454547e-06, "num_tokens": 2024185.0, "completions/mean_length": 257.0, "completions/min_length": 257.0, "completions/max_length": 257.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 257.0, "completions/min_terminated_length": 257.0, "completions/max_terminated_length": 257.0, "rewards/meter/mean": 0.998389720916748, "rewards/meter/std": 3.239075522287749e-05, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8557626008987427, "rewards/total_composite/std": 2.776351902866736e-05, "reward": 0.8557626008987427, "reward_std": 2.7763500838773325e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0011284128995612264, "sampling/sampling_logp_difference/max": 0.7351179122924805, "sampling/importance_sampling_ratio/min": 0.47944894433021545, "sampling/importance_sampling_ratio/mean": 0.9999052286148071, "sampling/importance_sampling_ratio/max": 1.2819808721542358, "entropy": 0.005354342225473374, "clip_ratio/low_mean": 0.00048638132284395397, "clip_ratio/low_min": 0.00048638132284395397, "clip_ratio/high_mean": 0.00048638132284395397, "clip_ratio/high_max": 0.00048638132284395397, "clip_ratio/region_mean": 0.0009727626456879079, "reward_total_mean": 0.8557626008987427, "reward_meter_mean": 0.998389720916748, "reward_meter_std": 3.239075522287749e-05, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8557626008987427, "reward_total_composite_std": 2.776351902866736e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 943.0} {"timestamp_utc": "2026-04-11T21:24:41Z", "mode": "train", "global_step": 944, "epoch": 0.0364535063330244, "loss": -0.0078, "grad_norm": 7.485598087310791, "learning_rate": 7.142424242424243e-06, "num_tokens": 2025988.0, "completions/mean_length": 72.375, "completions/min_length": 62.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.375, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.7796871066093445, "rewards/meter/std": 0.34613338112831116, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6995006799697876, "rewards/total_composite/std": 0.32923415303230286, "reward": 0.6995006799697876, "reward_std": 0.32923418283462524, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01891050860285759, "sampling/sampling_logp_difference/max": 1.1464146375656128, "sampling/importance_sampling_ratio/min": 0.31777405738830566, "sampling/importance_sampling_ratio/mean": 0.9990626573562622, "sampling/importance_sampling_ratio/max": 1.4766201972961426, "entropy": 0.08563010394573212, "clip_ratio/low_mean": 0.011398441973142326, "clip_ratio/low_min": 0.011398441973142326, "clip_ratio/high_mean": 0.011330341920256615, "clip_ratio/high_max": 0.011330341920256615, "clip_ratio/region_mean": 0.02272878389339894, "reward_total_mean": 0.6995006799697876, "reward_meter_mean": 0.7796871066093445, "reward_meter_std": 0.34613338112831116, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6995006799697876, "reward_total_composite_std": 0.32923415303230286, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 944.0} {"timestamp_utc": "2026-04-11T21:24:51Z", "mode": "train", "global_step": 945, "epoch": 0.03649212233549583, "loss": 0.0159, "grad_norm": 0.8819695711135864, "learning_rate": 7.1393939393939405e-06, "num_tokens": 2031484.0, "completions/mean_length": 453.0, "completions/min_length": 433.0, "completions/max_length": 465.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 453.0, "completions/min_terminated_length": 433.0, "completions/max_terminated_length": 465.0, "rewards/meter/mean": 0.9983506202697754, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.932692289352417, "rewards/count_adherence/std": 0.027196412906050682, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9311539530754089, "rewards/total_composite/std": 0.027151549234986305, "reward": 0.9311539530754089, "reward_std": 0.02715154178440571, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0009042817982845008, "sampling/sampling_logp_difference/max": 0.7498055696487427, "sampling/importance_sampling_ratio/min": 0.4724584221839905, "sampling/importance_sampling_ratio/mean": 1.0003540515899658, "sampling/importance_sampling_ratio/max": 1.8256886005401611, "entropy": 0.0033808891603257507, "clip_ratio/low_mean": 0.001094427308999002, "clip_ratio/low_min": 0.001094427308999002, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.001094427308999002, "reward_total_mean": 0.9311539530754089, "reward_meter_mean": 0.9983506202697754, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.932692289352417, "reward_count_adherence_std": 0.027196412906050682, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9311539530754089, "reward_total_composite_std": 0.027151549234986305, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 945.0} {"timestamp_utc": "2026-04-11T21:24:56Z", "mode": "train", "global_step": 946, "epoch": 0.03653073833796725, "loss": -0.0347, "grad_norm": 4.058627128601074, "learning_rate": 7.136363636363637e-06, "num_tokens": 2033843.0, "completions/mean_length": 112.875, "completions/min_length": 107.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.875, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.8164317607879639, "rewards/meter/std": 0.22834520041942596, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8164317607879639, "rewards/total_composite/std": 0.22834520041942596, "reward": 0.8164317607879639, "reward_std": 0.22834518551826477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012029947713017464, "sampling/sampling_logp_difference/max": 2.3145103454589844, "sampling/importance_sampling_ratio/min": 0.09881455451250076, "sampling/importance_sampling_ratio/mean": 1.0031535625457764, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03025160782271996, "clip_ratio/low_mean": 0.008112668758258224, "clip_ratio/low_min": 0.008112668758258224, "clip_ratio/high_mean": 0.00327742169611156, "clip_ratio/high_max": 0.00327742169611156, "clip_ratio/region_mean": 0.011390090454369783, "reward_total_mean": 0.8164317607879639, "reward_meter_mean": 0.8164317607879639, "reward_meter_std": 0.22834520041942596, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8164317607879639, "reward_total_composite_std": 0.22834520041942596, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 946.0} {"timestamp_utc": "2026-04-11T21:25:01Z", "mode": "train", "global_step": 947, "epoch": 0.036569354340438676, "loss": -0.0752, "grad_norm": 0.26604312658309937, "learning_rate": 7.133333333333334e-06, "num_tokens": 2035652.0, "completions/mean_length": 63.125, "completions/min_length": 49.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9983912706375122, "rewards/meter/std": 0.00011503982386784628, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9359943866729736, "rewards/total_composite/std": 0.17650160193443298, "reward": 0.9359943866729736, "reward_std": 0.1765015870332718, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005590436980128288, "sampling/sampling_logp_difference/max": 1.8964581489562988, "sampling/importance_sampling_ratio/min": 0.7738470435142517, "sampling/importance_sampling_ratio/mean": 1.0018616914749146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.010840466886293143, "clip_ratio/low_mean": 0.0025510203558951616, "clip_ratio/low_min": 0.0025510203558951616, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/region_mean": 0.006338899256661534, "reward_total_mean": 0.9359943866729736, "reward_meter_mean": 0.9983912706375122, "reward_meter_std": 0.00011503982386784628, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9359943866729736, "reward_total_composite_std": 0.17650160193443298, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 947.0} {"timestamp_utc": "2026-04-11T21:25:06Z", "mode": "train", "global_step": 948, "epoch": 0.0366079703429101, "loss": 0.0179, "grad_norm": 4.861557483673096, "learning_rate": 7.130303030303031e-06, "num_tokens": 2037366.0, "completions/mean_length": 59.25, "completions/min_length": 58.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.972287654876709, "rewards/meter/std": 0.028455156832933426, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.972287654876709, "rewards/total_composite/std": 0.028455156832933426, "reward": 0.972287654876709, "reward_std": 0.028455153107643127, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01554068736732006, "sampling/sampling_logp_difference/max": 2.139392852783203, "sampling/importance_sampling_ratio/min": 0.11772630363702774, "sampling/importance_sampling_ratio/mean": 0.9974233508110046, "sampling/importance_sampling_ratio/max": 1.2511993646621704, "entropy": 0.0650497677270323, "clip_ratio/low_mean": 0.0019841270986944437, "clip_ratio/low_min": 0.0019841270986944437, "clip_ratio/high_mean": 0.006465517217293382, "clip_ratio/high_max": 0.006465517217293382, "clip_ratio/region_mean": 0.008449644315987825, "reward_total_mean": 0.972287654876709, "reward_meter_mean": 0.972287654876709, "reward_meter_std": 0.028455156832933426, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.972287654876709, "reward_total_composite_std": 0.028455156832933426, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 948.0} {"timestamp_utc": "2026-04-11T21:25:11Z", "mode": "train", "global_step": 949, "epoch": 0.036646586345381524, "loss": 0.0193, "grad_norm": 2.2319347858428955, "learning_rate": 7.127272727272728e-06, "num_tokens": 2039035.0, "completions/mean_length": 59.625, "completions/min_length": 57.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.625, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8000860214233398, "rewards/meter/std": 0.002642143750563264, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8000860214233398, "rewards/total_composite/std": 0.002642143750563264, "reward": 0.8000860214233398, "reward_std": 0.002642149804159999, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005492615047842264, "sampling/sampling_logp_difference/max": 1.248208999633789, "sampling/importance_sampling_ratio/min": 0.28701838850975037, "sampling/importance_sampling_ratio/mean": 0.9990896582603455, "sampling/importance_sampling_ratio/max": 1.0505874156951904, "entropy": 0.015080863260664046, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0021929824724793434, "clip_ratio/high_max": 0.0021929824724793434, "clip_ratio/region_mean": 0.0021929824724793434, "reward_total_mean": 0.8000860214233398, "reward_meter_mean": 0.8000860214233398, "reward_meter_std": 0.002642143750563264, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8000860214233398, "reward_total_composite_std": 0.002642143750563264, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 949.0} {"timestamp_utc": "2026-04-11T21:25:17Z", "mode": "train", "global_step": 950, "epoch": 0.03668520234785295, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.124242424242424e-06, "num_tokens": 2042475.0, "completions/mean_length": 226.0, "completions/min_length": 226.0, "completions/max_length": 226.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 226.0, "completions/min_terminated_length": 226.0, "completions/max_terminated_length": 226.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00015113726840354502, "sampling/sampling_logp_difference/max": 0.03121185302734375, "sampling/importance_sampling_ratio/min": 0.9692702293395996, "sampling/importance_sampling_ratio/mean": 1.0001057386398315, "sampling/importance_sampling_ratio/max": 1.0076758861541748, "entropy": 0.0011664859484881163, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 950.0} {"timestamp_utc": "2026-04-11T21:26:32Z", "mode": "eval", "global_step": 950, "epoch": 0.03668520234785295, "eval_loss": NaN, "eval_runtime": 74.0905, "eval_samples_per_second": 1.404, "eval_steps_per_second": 0.175, "eval_num_tokens": 2042475.0, "eval_completions/mean_length": 188.81730769230768, "eval_completions/min_length": 51.76923076923077, "eval_completions/max_length": 384.84615384615387, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/mean_terminated_length": 185.52197852501502, "eval_completions/min_terminated_length": 51.76923076923077, "eval_completions/max_terminated_length": 371.6923076923077, "eval_rewards/meter/mean": 0.7363401467983539, "eval_rewards/meter/std": 0.37061490691625154, "eval_rewards/count_adherence/mean": 0.9601572018403274, "eval_rewards/count_adherence/std": 0.06895698721592243, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/total_composite/mean": 0.7043078862703763, "eval_rewards/total_composite/std": 0.3636963791572131, "eval_reward": 0.7043078862703763, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.004142561936392807, "eval_sampling/sampling_logp_difference/max": 0.8967751906468318, "eval_sampling/importance_sampling_ratio/min": 0.4854647425504831, "eval_sampling/importance_sampling_ratio/mean": 1.0009300938019385, "eval_sampling/importance_sampling_ratio/max": 1.3790802222031813, "eval_entropy": 0.042981731418806776, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7043078862703763, "eval_reward_meter_mean": 0.7363401467983539, "eval_reward_meter_std": 0.37061490691625154, "eval_reward_count_adherence_mean": 0.9601572018403274, "eval_reward_count_adherence_std": 0.06895698721592243, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_total_composite_mean": 0.7043078862703763, "eval_reward_total_composite_std": 0.3636963791572131, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 950.0} {"timestamp_utc": "2026-04-11T21:26:39Z", "mode": "train", "global_step": 951, "epoch": 0.03672381835032437, "loss": -0.0151, "grad_norm": 3.191176414489746, "learning_rate": 7.121212121212122e-06, "num_tokens": 2044368.0, "completions/mean_length": 76.625, "completions/min_length": 71.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.625, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.22621294856071472, "rewards/meter/std": 0.13094781339168549, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.22621294856071472, "rewards/total_composite/std": 0.13094781339168549, "reward": 0.22621294856071472, "reward_std": 0.13094781339168549, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02579626999795437, "sampling/sampling_logp_difference/max": 2.398671865463257, "sampling/importance_sampling_ratio/min": 0.09083851426839828, "sampling/importance_sampling_ratio/mean": 0.9951958656311035, "sampling/importance_sampling_ratio/max": 1.6687957048416138, "entropy": 0.049755130894482136, "clip_ratio/low_mean": 0.008378971251659095, "clip_ratio/low_min": 0.008378971251659095, "clip_ratio/high_mean": 0.0030304142273962498, "clip_ratio/high_max": 0.0030304142273962498, "clip_ratio/region_mean": 0.011409385479055345, "reward_total_mean": 0.22621294856071472, "reward_meter_mean": 0.22621294856071472, "reward_meter_std": 0.13094781339168549, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.22621294856071472, "reward_total_composite_std": 0.13094781339168549, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 951.0} {"timestamp_utc": "2026-04-11T21:26:46Z", "mode": "train", "global_step": 952, "epoch": 0.036762434352795796, "loss": -0.0001, "grad_norm": 0.08681128174066544, "learning_rate": 7.118181818181819e-06, "num_tokens": 2046577.0, "completions/mean_length": 97.125, "completions/min_length": 97.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.125, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9983540177345276, "rewards/meter/std": 9.609481821826193e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9983540177345276, "rewards/total_composite/std": 9.609481821826193e-06, "reward": 0.9983540177345276, "reward_std": 9.609481821826193e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0018863894511014223, "sampling/sampling_logp_difference/max": 0.6860620975494385, "sampling/importance_sampling_ratio/min": 0.503555178642273, "sampling/importance_sampling_ratio/mean": 0.9998738765716553, "sampling/importance_sampling_ratio/max": 1.0550848245620728, "entropy": 0.005472837190609425, "clip_ratio/low_mean": 0.002577319508418441, "clip_ratio/low_min": 0.002577319508418441, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.002577319508418441, "reward_total_mean": 0.9983540177345276, "reward_meter_mean": 0.9983540177345276, "reward_meter_std": 9.609481821826193e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9983540177345276, "reward_total_composite_std": 9.609481821826193e-06, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 952.0} {"timestamp_utc": "2026-04-11T21:26:53Z", "mode": "train", "global_step": 953, "epoch": 0.03680105035526722, "loss": -0.0018, "grad_norm": 2.79339861869812, "learning_rate": 7.115151515151516e-06, "num_tokens": 2048642.0, "completions/mean_length": 101.125, "completions/min_length": 98.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.125, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.7972941398620605, "rewards/meter/std": 0.191716268658638, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7972941398620605, "rewards/total_composite/std": 0.191716268658638, "reward": 0.7972941398620605, "reward_std": 0.1917162537574768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023367850109934807, "sampling/sampling_logp_difference/max": 1.63095223903656, "sampling/importance_sampling_ratio/min": 0.19574308395385742, "sampling/importance_sampling_ratio/mean": 0.9967895150184631, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07592712761834264, "clip_ratio/low_mean": 0.0050633890787139535, "clip_ratio/low_min": 0.0050633890787139535, "clip_ratio/high_mean": 0.013675765483640134, "clip_ratio/high_max": 0.013675765483640134, "clip_ratio/region_mean": 0.018739154562354088, "reward_total_mean": 0.7972941398620605, "reward_meter_mean": 0.7972941398620605, "reward_meter_std": 0.191716268658638, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7972941398620605, "reward_total_composite_std": 0.191716268658638, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 953.0} {"timestamp_utc": "2026-04-11T21:26:58Z", "mode": "train", "global_step": 954, "epoch": 0.036839666357738644, "loss": 0.0521, "grad_norm": 7.48327112197876, "learning_rate": 7.1121212121212125e-06, "num_tokens": 2050226.0, "completions/mean_length": 55.0, "completions/min_length": 50.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.6734884977340698, "rewards/meter/std": 0.35674989223480225, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6734884977340698, "rewards/total_composite/std": 0.35674989223480225, "reward": 0.6734884977340698, "reward_std": 0.35674986243247986, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03522590547800064, "sampling/sampling_logp_difference/max": 2.2291746139526367, "sampling/importance_sampling_ratio/min": 0.36284998059272766, "sampling/importance_sampling_ratio/mean": 1.0048383474349976, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10999267641454935, "clip_ratio/low_mean": 0.012973579345270991, "clip_ratio/low_min": 0.012973579345270991, "clip_ratio/high_mean": 0.01917571760714054, "clip_ratio/high_max": 0.01917571760714054, "clip_ratio/region_mean": 0.03214929695241153, "reward_total_mean": 0.6734884977340698, "reward_meter_mean": 0.6734884977340698, "reward_meter_std": 0.35674989223480225, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6734884977340698, "reward_total_composite_std": 0.35674989223480225, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 954.0} {"timestamp_utc": "2026-04-11T21:27:02Z", "mode": "train", "global_step": 955, "epoch": 0.03687828236021007, "loss": -0.0155, "grad_norm": 5.164478778839111, "learning_rate": 7.10909090909091e-06, "num_tokens": 2051906.0, "completions/mean_length": 52.0, "completions/min_length": 50.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9581389427185059, "rewards/meter/std": 0.005824621766805649, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9581389427185059, "rewards/total_composite/std": 0.005824621766805649, "reward": 0.9581389427185059, "reward_std": 0.005824611056596041, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03934483602643013, "sampling/sampling_logp_difference/max": 2.24235200881958, "sampling/importance_sampling_ratio/min": 0.10620840638875961, "sampling/importance_sampling_ratio/mean": 0.9873109459877014, "sampling/importance_sampling_ratio/max": 1.281923770904541, "entropy": 0.104397336486727, "clip_ratio/low_mean": 0.0024509804788976908, "clip_ratio/low_min": 0.0024509804788976908, "clip_ratio/high_mean": 0.02363362116739154, "clip_ratio/high_max": 0.02363362116739154, "clip_ratio/region_mean": 0.02608460164628923, "reward_total_mean": 0.9581389427185059, "reward_meter_mean": 0.9581389427185059, "reward_meter_std": 0.005824621766805649, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9581389427185059, "reward_total_composite_std": 0.005824621766805649, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 955.0} {"timestamp_utc": "2026-04-11T21:27:08Z", "mode": "train", "global_step": 956, "epoch": 0.03691689836268149, "loss": -0.0132, "grad_norm": 2.844005584716797, "learning_rate": 7.106060606060606e-06, "num_tokens": 2054413.0, "completions/mean_length": 143.375, "completions/min_length": 136.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.375, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.21800079941749573, "rewards/meter/std": 0.2732958495616913, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.2139286994934082, "rewards/total_composite/std": 0.27560707926750183, "reward": 0.2139286994934082, "reward_std": 0.27560704946517944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018258651718497276, "sampling/sampling_logp_difference/max": 6.802255153656006, "sampling/importance_sampling_ratio/min": 0.0011112663196399808, "sampling/importance_sampling_ratio/mean": 0.9955170154571533, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02346757787745446, "clip_ratio/low_mean": 0.008840867376420647, "clip_ratio/low_min": 0.008840867376420647, "clip_ratio/high_mean": 0.0016891892300918698, "clip_ratio/high_max": 0.0016891892300918698, "clip_ratio/region_mean": 0.010530056606512517, "reward_total_mean": 0.2139286994934082, "reward_meter_mean": 0.21800079941749573, "reward_meter_std": 0.2732958495616913, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.2139286994934082, "reward_total_composite_std": 0.27560707926750183, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 956.0} {"timestamp_utc": "2026-04-11T21:27:17Z", "mode": "train", "global_step": 957, "epoch": 0.03695551436515292, "loss": -0.0182, "grad_norm": 1.7913213968276978, "learning_rate": 7.103030303030304e-06, "num_tokens": 2058436.0, "completions/mean_length": 304.875, "completions/min_length": 290.0, "completions/max_length": 307.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 304.875, "completions/min_terminated_length": 290.0, "completions/max_terminated_length": 307.0, "rewards/meter/mean": 0.8853950500488281, "rewards/meter/std": 0.3146956264972687, "rewards/count_adherence/mean": 0.887499988079071, "rewards/count_adherence/std": 0.035355325788259506, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7955231666564941, "rewards/total_composite/std": 0.28699445724487305, "reward": 0.7955231666564941, "reward_std": 0.28699445724487305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004229568410664797, "sampling/sampling_logp_difference/max": 1.8866181373596191, "sampling/importance_sampling_ratio/min": 0.15158358216285706, "sampling/importance_sampling_ratio/mean": 0.9997403621673584, "sampling/importance_sampling_ratio/max": 1.6602782011032104, "entropy": 0.0158452526666224, "clip_ratio/low_mean": 0.0008620689623057842, "clip_ratio/low_min": 0.0008620689623057842, "clip_ratio/high_mean": 0.0012214983289595693, "clip_ratio/high_max": 0.0012214983289595693, "clip_ratio/region_mean": 0.0020835672912653536, "reward_total_mean": 0.7955231666564941, "reward_meter_mean": 0.8853950500488281, "reward_meter_std": 0.3146956264972687, "reward_count_adherence_mean": 0.887499988079071, "reward_count_adherence_std": 0.035355325788259506, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7955231666564941, "reward_total_composite_std": 0.28699445724487305, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 957.0} {"timestamp_utc": "2026-04-11T21:27:23Z", "mode": "train", "global_step": 958, "epoch": 0.03699413036762434, "loss": -0.0356, "grad_norm": 4.510936260223389, "learning_rate": 7.100000000000001e-06, "num_tokens": 2060248.0, "completions/mean_length": 59.5, "completions/min_length": 48.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.5, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.8756954669952393, "rewards/meter/std": 0.1648823469877243, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8756954669952393, "rewards/total_composite/std": 0.1648823469877243, "reward": 0.8756954669952393, "reward_std": 0.1648823320865631, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040581345558166504, "sampling/sampling_logp_difference/max": 5.4065842628479, "sampling/importance_sampling_ratio/min": 0.004486940335482359, "sampling/importance_sampling_ratio/mean": 1.0015078783035278, "sampling/importance_sampling_ratio/max": 1.9976530075073242, "entropy": 0.05970208114013076, "clip_ratio/low_mean": 0.004759339150041342, "clip_ratio/low_min": 0.004759339150041342, "clip_ratio/high_mean": 0.008196721319109201, "clip_ratio/high_max": 0.008196721319109201, "clip_ratio/region_mean": 0.012956060469150543, "reward_total_mean": 0.8756954669952393, "reward_meter_mean": 0.8756954669952393, "reward_meter_std": 0.1648823469877243, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8756954669952393, "reward_total_composite_std": 0.1648823469877243, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 958.0} {"timestamp_utc": "2026-04-11T21:27:29Z", "mode": "train", "global_step": 959, "epoch": 0.037032746370095765, "loss": -0.0045, "grad_norm": 4.195918560028076, "learning_rate": 7.096969696969698e-06, "num_tokens": 2061977.0, "completions/mean_length": 58.125, "completions/min_length": 52.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9801746606826782, "rewards/meter/std": 0.020054606720805168, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9801746606826782, "rewards/total_composite/std": 0.020054606720805168, "reward": 0.9801746606826782, "reward_std": 0.02005460113286972, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014512870460748672, "sampling/sampling_logp_difference/max": 0.8872404098510742, "sampling/importance_sampling_ratio/min": 0.41179054975509644, "sampling/importance_sampling_ratio/mean": 1.0003046989440918, "sampling/importance_sampling_ratio/max": 1.7301915884017944, "entropy": 0.028893944574519992, "clip_ratio/low_mean": 0.00657894741743803, "clip_ratio/low_min": 0.00657894741743803, "clip_ratio/high_mean": 0.006856872700154781, "clip_ratio/high_max": 0.006856872700154781, "clip_ratio/region_mean": 0.013435820117592812, "reward_total_mean": 0.9801746606826782, "reward_meter_mean": 0.9801746606826782, "reward_meter_std": 0.020054606720805168, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9801746606826782, "reward_total_composite_std": 0.020054606720805168, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 959.0} {"timestamp_utc": "2026-04-11T21:27:34Z", "mode": "train", "global_step": 960, "epoch": 0.03707136237256719, "loss": 0.0012, "grad_norm": 2.899587631225586, "learning_rate": 7.093939393939394e-06, "num_tokens": 2063723.0, "completions/mean_length": 59.25, "completions/min_length": 55.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.1531568318605423, "rewards/meter/std": 0.12182258069515228, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.1531568318605423, "rewards/total_composite/std": 0.12182258069515228, "reward": 0.1531568318605423, "reward_std": 0.12182258814573288, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06045001745223999, "sampling/sampling_logp_difference/max": 3.3967068195343018, "sampling/importance_sampling_ratio/min": 0.0334833562374115, "sampling/importance_sampling_ratio/mean": 0.9934316277503967, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10416628792881966, "clip_ratio/low_mean": 0.015023555606603622, "clip_ratio/low_min": 0.015023555606603622, "clip_ratio/high_mean": 0.014689265750348568, "clip_ratio/high_max": 0.014689265750348568, "clip_ratio/region_mean": 0.02971282135695219, "reward_total_mean": 0.1531568318605423, "reward_meter_mean": 0.1531568318605423, "reward_meter_std": 0.12182258069515228, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.1531568318605423, "reward_total_composite_std": 0.12182258069515228, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 960.0} {"timestamp_utc": "2026-04-11T21:27:39Z", "mode": "train", "global_step": 961, "epoch": 0.03710997837503861, "loss": 0.0179, "grad_norm": 2.7200634479522705, "learning_rate": 7.0909090909090916e-06, "num_tokens": 2065590.0, "completions/mean_length": 58.375, "completions/min_length": 54.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9704042673110962, "rewards/meter/std": 0.03516518697142601, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9704042673110962, "rewards/total_composite/std": 0.03516518697142601, "reward": 0.9704042673110962, "reward_std": 0.03516519442200661, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022747451439499855, "sampling/sampling_logp_difference/max": 2.6428117752075195, "sampling/importance_sampling_ratio/min": 0.07116090506315231, "sampling/importance_sampling_ratio/mean": 0.9974169731140137, "sampling/importance_sampling_ratio/max": 1.5621143579483032, "entropy": 0.06281700287945569, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/high_mean": 0.004433458903804421, "clip_ratio/high_max": 0.004433458903804421, "clip_ratio/region_mean": 0.008465716848149896, "reward_total_mean": 0.9704042673110962, "reward_meter_mean": 0.9704042673110962, "reward_meter_std": 0.03516518697142601, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9704042673110962, "reward_total_composite_std": 0.03516518697142601, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 961.0} {"timestamp_utc": "2026-04-11T21:27:45Z", "mode": "train", "global_step": 962, "epoch": 0.03714859437751004, "loss": 0.0385, "grad_norm": 2.241823196411133, "learning_rate": 7.087878787878788e-06, "num_tokens": 2067752.0, "completions/mean_length": 100.25, "completions/min_length": 97.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.25, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.8679694533348083, "rewards/meter/std": 0.3398498296737671, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8679694533348083, "rewards/total_composite/std": 0.3398498296737671, "reward": 0.8679694533348083, "reward_std": 0.3398498296737671, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010863143019378185, "sampling/sampling_logp_difference/max": 1.2088133096694946, "sampling/importance_sampling_ratio/min": 0.2985513508319855, "sampling/importance_sampling_ratio/mean": 1.0026658773422241, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05559232854284346, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.005115979234687984, "clip_ratio/high_max": 0.005115979234687984, "clip_ratio/region_mean": 0.008494357694871724, "reward_total_mean": 0.8679694533348083, "reward_meter_mean": 0.8679694533348083, "reward_meter_std": 0.3398498296737671, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8679694533348083, "reward_total_composite_std": 0.3398498296737671, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 962.0} {"timestamp_utc": "2026-04-11T21:27:49Z", "mode": "train", "global_step": 963, "epoch": 0.03718721037998146, "loss": 0.0019, "grad_norm": 5.761496067047119, "learning_rate": 7.084848484848485e-06, "num_tokens": 2069398.0, "completions/mean_length": 25.75, "completions/min_length": 25.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.75, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.8386284708976746, "rewards/meter/std": 0.26306402683258057, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8386284708976746, "rewards/total_composite/std": 0.26306402683258057, "reward": 0.8386284708976746, "reward_std": 0.26306402683258057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03292413800954819, "sampling/sampling_logp_difference/max": 1.6141786575317383, "sampling/importance_sampling_ratio/min": 0.1990540772676468, "sampling/importance_sampling_ratio/mean": 1.0044740438461304, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10780723765492439, "clip_ratio/low_mean": 0.009999999776482582, "clip_ratio/low_min": 0.009999999776482582, "clip_ratio/high_mean": 0.019807692151516676, "clip_ratio/high_max": 0.019807692151516676, "clip_ratio/region_mean": 0.029807691927999258, "reward_total_mean": 0.8386284708976746, "reward_meter_mean": 0.8386284708976746, "reward_meter_std": 0.26306402683258057, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8386284708976746, "reward_total_composite_std": 0.26306402683258057, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 963.0} {"timestamp_utc": "2026-04-11T21:27:54Z", "mode": "train", "global_step": 964, "epoch": 0.037225826382452885, "loss": -0.0039, "grad_norm": 4.339460849761963, "learning_rate": 7.081818181818182e-06, "num_tokens": 2070962.0, "completions/mean_length": 33.5, "completions/min_length": 33.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9982439875602722, "rewards/meter/std": 0.000402480160119012, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9982439875602722, "rewards/total_composite/std": 0.000402480160119012, "reward": 0.9982439875602722, "reward_std": 0.000402480160119012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011125919409096241, "sampling/sampling_logp_difference/max": 0.5592041015625, "sampling/importance_sampling_ratio/min": 0.5716638565063477, "sampling/importance_sampling_ratio/mean": 0.9981675744056702, "sampling/importance_sampling_ratio/max": 1.2405402660369873, "entropy": 0.03733847592957318, "clip_ratio/low_mean": 0.011029412038624287, "clip_ratio/low_min": 0.011029412038624287, "clip_ratio/high_mean": 0.007352941203862429, "clip_ratio/high_max": 0.007352941203862429, "clip_ratio/region_mean": 0.018382353242486715, "reward_total_mean": 0.9982439875602722, "reward_meter_mean": 0.9982439875602722, "reward_meter_std": 0.000402480160119012, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9982439875602722, "reward_total_composite_std": 0.000402480160119012, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 964.0} {"timestamp_utc": "2026-04-11T21:28:02Z", "mode": "train", "global_step": 965, "epoch": 0.03726444238492431, "loss": -0.0007, "grad_norm": 0.3006925880908966, "learning_rate": 7.07878787878788e-06, "num_tokens": 2075457.0, "completions/mean_length": 355.875, "completions/min_length": 341.0, "completions/max_length": 358.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 355.875, "completions/min_terminated_length": 341.0, "completions/max_terminated_length": 358.0, "rewards/meter/mean": 0.9899473786354065, "rewards/meter/std": 0.0003307900333311409, "rewards/count_adherence/mean": 0.9090909361839294, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8999521732330322, "rewards/total_composite/std": 0.0003007233899552375, "reward": 0.8999521732330322, "reward_std": 0.0003007233899552375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004097128286957741, "sampling/sampling_logp_difference/max": 2.5838499069213867, "sampling/importance_sampling_ratio/min": 0.07548283785581589, "sampling/importance_sampling_ratio/mean": 0.9990352988243103, "sampling/importance_sampling_ratio/max": 1.4331867694854736, "entropy": 0.005794623983092606, "clip_ratio/low_mean": 0.0017458100046496838, "clip_ratio/low_min": 0.0017458100046496838, "clip_ratio/high_mean": 0.00034916200092993677, "clip_ratio/high_max": 0.00034916200092993677, "clip_ratio/region_mean": 0.0020949720055796206, "reward_total_mean": 0.8999521732330322, "reward_meter_mean": 0.9899473786354065, "reward_meter_std": 0.0003307900333311409, "reward_count_adherence_mean": 0.9090909361839294, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8999521732330322, "reward_total_composite_std": 0.0003007233899552375, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 965.0} {"timestamp_utc": "2026-04-11T21:28:07Z", "mode": "train", "global_step": 966, "epoch": 0.037303058387395734, "loss": -0.0079, "grad_norm": 2.565558433532715, "learning_rate": 7.075757575757576e-06, "num_tokens": 2077321.0, "completions/mean_length": 66.0, "completions/min_length": 65.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9984169602394104, "rewards/meter/std": 0.00012341170804575086, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984169602394104, "rewards/total_composite/std": 0.00012341170804575086, "reward": 0.9984169602394104, "reward_std": 0.00012341169349383563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006317383609712124, "sampling/sampling_logp_difference/max": 0.6659119725227356, "sampling/importance_sampling_ratio/min": 0.5138047337532043, "sampling/importance_sampling_ratio/mean": 0.9997684955596924, "sampling/importance_sampling_ratio/max": 1.2385183572769165, "entropy": 0.02148077730089426, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0018115942366421223, "clip_ratio/high_max": 0.0018115942366421223, "clip_ratio/region_mean": 0.0018115942366421223, "reward_total_mean": 0.9984169602394104, "reward_meter_mean": 0.9984169602394104, "reward_meter_std": 0.00012341170804575086, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984169602394104, "reward_total_composite_std": 0.00012341170804575086, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 966.0} {"timestamp_utc": "2026-04-11T21:28:13Z", "mode": "train", "global_step": 967, "epoch": 0.03734167438986716, "loss": -0.0093, "grad_norm": 1.5171047449111938, "learning_rate": 7.072727272727273e-06, "num_tokens": 2079135.0, "completions/mean_length": 58.75, "completions/min_length": 58.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9833565950393677, "rewards/meter/std": 0.0013543061213567853, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9833565950393677, "rewards/total_composite/std": 0.0013543061213567853, "reward": 0.9833565950393677, "reward_std": 0.0013542849337682128, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006052408833056688, "sampling/sampling_logp_difference/max": 0.32959139347076416, "sampling/importance_sampling_ratio/min": 0.7192175984382629, "sampling/importance_sampling_ratio/mean": 1.0012885332107544, "sampling/importance_sampling_ratio/max": 1.24648118019104, "entropy": 0.04555001808330417, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0042372881434857845, "reward_total_mean": 0.9833565950393677, "reward_meter_mean": 0.9833565950393677, "reward_meter_std": 0.0013543061213567853, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9833565950393677, "reward_total_composite_std": 0.0013543061213567853, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 967.0} {"timestamp_utc": "2026-04-11T21:28:18Z", "mode": "train", "global_step": 968, "epoch": 0.03738029039233858, "loss": 0.1033, "grad_norm": 7.06599235534668, "learning_rate": 7.06969696969697e-06, "num_tokens": 2081078.0, "completions/mean_length": 58.875, "completions/min_length": 52.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.6837669610977173, "rewards/meter/std": 0.2890775203704834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6837669610977173, "rewards/total_composite/std": 0.2890775203704834, "reward": 0.6837669610977173, "reward_std": 0.2890775203704834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018488025292754173, "sampling/sampling_logp_difference/max": 1.4911084175109863, "sampling/importance_sampling_ratio/min": 0.22512298822402954, "sampling/importance_sampling_ratio/mean": 0.9969645142555237, "sampling/importance_sampling_ratio/max": 1.7272429466247559, "entropy": 0.07422188855707645, "clip_ratio/low_mean": 0.0024038462433964014, "clip_ratio/low_min": 0.0024038462433964014, "clip_ratio/high_mean": 0.008370535913854837, "clip_ratio/high_max": 0.008370535913854837, "clip_ratio/region_mean": 0.010774382157251239, "reward_total_mean": 0.6837669610977173, "reward_meter_mean": 0.6837669610977173, "reward_meter_std": 0.2890775203704834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6837669610977173, "reward_total_composite_std": 0.2890775203704834, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 968.0} {"timestamp_utc": "2026-04-11T21:28:23Z", "mode": "train", "global_step": 969, "epoch": 0.037418906394810006, "loss": -0.0135, "grad_norm": 2.1965885162353516, "learning_rate": 7.066666666666667e-06, "num_tokens": 2082761.0, "completions/mean_length": 57.375, "completions/min_length": 55.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.375, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9937889575958252, "rewards/meter/std": 0.005989787634462118, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9937889575958252, "rewards/total_composite/std": 0.005989787634462118, "reward": 0.9937889575958252, "reward_std": 0.005989796482026577, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027677176520228386, "sampling/sampling_logp_difference/max": 3.6823887825012207, "sampling/importance_sampling_ratio/min": 0.025162795558571815, "sampling/importance_sampling_ratio/mean": 0.9993962049484253, "sampling/importance_sampling_ratio/max": 1.6849017143249512, "entropy": 0.09074101317673922, "clip_ratio/low_mean": 0.0066964286379516125, "clip_ratio/low_min": 0.0066964286379516125, "clip_ratio/high_mean": 0.00870644859969616, "clip_ratio/high_max": 0.00870644859969616, "clip_ratio/region_mean": 0.015402877237647772, "reward_total_mean": 0.9937889575958252, "reward_meter_mean": 0.9937889575958252, "reward_meter_std": 0.005989787634462118, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9937889575958252, "reward_total_composite_std": 0.005989787634462118, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 969.0} {"timestamp_utc": "2026-04-11T21:28:29Z", "mode": "train", "global_step": 970, "epoch": 0.03745752239728143, "loss": -0.0344, "grad_norm": 1.6494258642196655, "learning_rate": 7.063636363636365e-06, "num_tokens": 2085430.0, "completions/mean_length": 128.625, "completions/min_length": 115.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.625, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9934996366500854, "rewards/meter/std": 0.0005216970457695425, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9624683856964111, "rewards/total_composite/std": 0.08797280490398407, "reward": 0.9624683856964111, "reward_std": 0.08797280490398407, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013765535317361355, "sampling/sampling_logp_difference/max": 1.773430347442627, "sampling/importance_sampling_ratio/min": 0.1697496771812439, "sampling/importance_sampling_ratio/mean": 0.9975191950798035, "sampling/importance_sampling_ratio/max": 1.741287112236023, "entropy": 0.03762633865699172, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.008610145130660385, "clip_ratio/high_max": 0.008610145130660385, "clip_ratio/region_mean": 0.008610145130660385, "reward_total_mean": 0.9624683856964111, "reward_meter_mean": 0.9934996366500854, "reward_meter_std": 0.0005216970457695425, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9624683856964111, "reward_total_composite_std": 0.08797280490398407, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 970.0} {"timestamp_utc": "2026-04-11T21:28:34Z", "mode": "train", "global_step": 971, "epoch": 0.037496138399752854, "loss": 0.0043, "grad_norm": 2.0139245986938477, "learning_rate": 7.060606060606061e-06, "num_tokens": 2087540.0, "completions/mean_length": 59.75, "completions/min_length": 58.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9961794018745422, "rewards/meter/std": 0.0001523538667242974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9961794018745422, "rewards/total_composite/std": 0.0001523538667242974, "reward": 0.9961794018745422, "reward_std": 0.00015233788872137666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0158246960490942, "sampling/sampling_logp_difference/max": 1.0432453155517578, "sampling/importance_sampling_ratio/min": 0.3523094654083252, "sampling/importance_sampling_ratio/mean": 1.003324270248413, "sampling/importance_sampling_ratio/max": 1.9049383401870728, "entropy": 0.08410773845389485, "clip_ratio/low_mean": 0.004171301377937198, "clip_ratio/low_min": 0.004171301377937198, "clip_ratio/high_mean": 0.0062511577270925045, "clip_ratio/high_max": 0.0062511577270925045, "clip_ratio/region_mean": 0.010422459105029702, "reward_total_mean": 0.9961794018745422, "reward_meter_mean": 0.9961794018745422, "reward_meter_std": 0.0001523538667242974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9961794018745422, "reward_total_composite_std": 0.0001523538667242974, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 971.0} {"timestamp_utc": "2026-04-11T21:28:40Z", "mode": "train", "global_step": 972, "epoch": 0.037534754402224285, "loss": -0.0003, "grad_norm": 0.06810560822486877, "learning_rate": 7.057575757575759e-06, "num_tokens": 2090428.0, "completions/mean_length": 181.0, "completions/min_length": 181.0, "completions/max_length": 181.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.0, "completions/min_terminated_length": 181.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.998572051525116, "rewards/meter/std": 5.664536729454994e-05, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7988576292991638, "rewards/total_composite/std": 4.530786827672273e-05, "reward": 0.7988576292991638, "reward_std": 4.531388549366966e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0007551187882199883, "sampling/sampling_logp_difference/max": 0.5210800170898438, "sampling/importance_sampling_ratio/min": 0.5938788056373596, "sampling/importance_sampling_ratio/mean": 1.000138521194458, "sampling/importance_sampling_ratio/max": 1.3649401664733887, "entropy": 0.0022229965252336115, "clip_ratio/low_mean": 0.0013812155229970813, "clip_ratio/low_min": 0.0013812155229970813, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0013812155229970813, "reward_total_mean": 0.7988576292991638, "reward_meter_mean": 0.998572051525116, "reward_meter_std": 5.664536729454994e-05, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7988576292991638, "reward_total_composite_std": 4.530786827672273e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 972.0} {"timestamp_utc": "2026-04-11T21:28:46Z", "mode": "train", "global_step": 973, "epoch": 0.03757337040469571, "loss": 0.0763, "grad_norm": 4.874419689178467, "learning_rate": 7.054545454545455e-06, "num_tokens": 2092967.0, "completions/mean_length": 124.375, "completions/min_length": 118.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.375, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.8988627791404724, "rewards/meter/std": 0.2758934497833252, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8609772324562073, "rewards/total_composite/std": 0.29556939005851746, "reward": 0.8609772324562073, "reward_std": 0.29556936025619507, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03274437040090561, "sampling/sampling_logp_difference/max": 1.4704504013061523, "sampling/importance_sampling_ratio/min": 0.22982195019721985, "sampling/importance_sampling_ratio/mean": 1.004819631576538, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2657974073663354, "clip_ratio/low_mean": 0.013400557916611433, "clip_ratio/low_min": 0.013400557916611433, "clip_ratio/high_mean": 0.007353089866228402, "clip_ratio/high_max": 0.007353089866228402, "clip_ratio/region_mean": 0.020753647782839835, "reward_total_mean": 0.8609772324562073, "reward_meter_mean": 0.8988627791404724, "reward_meter_std": 0.2758934497833252, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8609772324562073, "reward_total_composite_std": 0.29556939005851746, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 973.0} {"timestamp_utc": "2026-04-11T21:28:52Z", "mode": "train", "global_step": 974, "epoch": 0.03761198640716713, "loss": -0.0236, "grad_norm": 3.239126205444336, "learning_rate": 7.0515151515151525e-06, "num_tokens": 2094480.0, "completions/mean_length": 50.125, "completions/min_length": 49.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.125, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.08261874318122864, "rewards/meter/std": 0.040867600589990616, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.08261874318122864, "rewards/total_composite/std": 0.040867600589990616, "reward": 0.08261874318122864, "reward_std": 0.040867604315280914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007985641248524189, "sampling/sampling_logp_difference/max": 1.2962665557861328, "sampling/importance_sampling_ratio/min": 0.27355116605758667, "sampling/importance_sampling_ratio/mean": 1.002482533454895, "sampling/importance_sampling_ratio/max": 1.2251895666122437, "entropy": 0.051621134858578444, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.08261874318122864, "reward_meter_mean": 0.08261874318122864, "reward_meter_std": 0.040867600589990616, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.08261874318122864, "reward_total_composite_std": 0.040867600589990616, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 974.0} {"timestamp_utc": "2026-04-11T21:28:57Z", "mode": "train", "global_step": 975, "epoch": 0.03765060240963856, "loss": 0.0348, "grad_norm": 3.62621808052063, "learning_rate": 7.048484848484849e-06, "num_tokens": 2096832.0, "completions/mean_length": 119.0, "completions/min_length": 114.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.0, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.007328447885811329, "rewards/meter/std": 0.002096648560836911, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.005862758494913578, "rewards/total_composite/std": 0.0016773188253864646, "reward": 0.005862758494913578, "reward_std": 0.0016773188253864646, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02160906046628952, "sampling/sampling_logp_difference/max": 2.983750820159912, "sampling/importance_sampling_ratio/min": 0.05060267820954323, "sampling/importance_sampling_ratio/mean": 1.0029712915420532, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.059714815113693476, "clip_ratio/low_mean": 0.0049509028904139996, "clip_ratio/low_min": 0.0049509028904139996, "clip_ratio/high_mean": 0.004201680887490511, "clip_ratio/high_max": 0.004201680887490511, "clip_ratio/region_mean": 0.00915258377790451, "reward_total_mean": 0.005862758494913578, "reward_meter_mean": 0.007328447885811329, "reward_meter_std": 0.002096648560836911, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.005862758494913578, "reward_total_composite_std": 0.0016773188253864646, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 975.0} {"timestamp_utc": "2026-04-11T21:29:02Z", "mode": "train", "global_step": 976, "epoch": 0.03768921841210998, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.045454545454546e-06, "num_tokens": 2098936.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00047403317876160145, "sampling/sampling_logp_difference/max": 0.0835278332233429, "sampling/importance_sampling_ratio/min": 0.9198654890060425, "sampling/importance_sampling_ratio/mean": 0.9998888373374939, "sampling/importance_sampling_ratio/max": 1.0297996997833252, "entropy": 0.004023725254228339, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 976.0} {"timestamp_utc": "2026-04-11T21:29:08Z", "mode": "train", "global_step": 977, "epoch": 0.037727834414581406, "loss": 0.0171, "grad_norm": 4.354089260101318, "learning_rate": 7.0424242424242426e-06, "num_tokens": 2101150.0, "completions/mean_length": 96.75, "completions/min_length": 88.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.75, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.6210145950317383, "rewards/meter/std": 0.5073013305664062, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6207942962646484, "rewards/total_composite/std": 0.5076071619987488, "reward": 0.6207942962646484, "reward_std": 0.507607102394104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02025706321001053, "sampling/sampling_logp_difference/max": 2.9257829189300537, "sampling/importance_sampling_ratio/min": 0.053622692823410034, "sampling/importance_sampling_ratio/mean": 0.9985988140106201, "sampling/importance_sampling_ratio/max": 1.6468206644058228, "entropy": 0.07154249120503664, "clip_ratio/low_mean": 0.00643545389175415, "clip_ratio/low_min": 0.00643545389175415, "clip_ratio/high_mean": 0.011793343583121896, "clip_ratio/high_max": 0.011793343583121896, "clip_ratio/region_mean": 0.018228797474876046, "reward_total_mean": 0.6207942962646484, "reward_meter_mean": 0.6210145950317383, "reward_meter_std": 0.5073013305664062, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6207942962646484, "reward_total_composite_std": 0.5076071619987488, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 977.0} {"timestamp_utc": "2026-04-11T21:29:14Z", "mode": "train", "global_step": 978, "epoch": 0.03776645041705283, "loss": -0.1073, "grad_norm": 1.7061519622802734, "learning_rate": 7.039393939393941e-06, "num_tokens": 2103297.0, "completions/mean_length": 96.375, "completions/min_length": 84.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.375, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.5528889894485474, "rewards/meter/std": 0.47514161467552185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5528889894485474, "rewards/total_composite/std": 0.47514161467552185, "reward": 0.5528889894485474, "reward_std": 0.47514158487319946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018114496022462845, "sampling/sampling_logp_difference/max": 6.2520365715026855, "sampling/importance_sampling_ratio/min": 0.0019265266600996256, "sampling/importance_sampling_ratio/mean": 1.0000500679016113, "sampling/importance_sampling_ratio/max": 1.5997724533081055, "entropy": 0.0397267104126513, "clip_ratio/low_mean": 0.005884740385226905, "clip_ratio/low_min": 0.005884740385226905, "clip_ratio/high_mean": 0.006993489572778344, "clip_ratio/high_max": 0.006993489572778344, "clip_ratio/region_mean": 0.01287822995800525, "reward_total_mean": 0.5528889894485474, "reward_meter_mean": 0.5528889894485474, "reward_meter_std": 0.47514161467552185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5528889894485474, "reward_total_composite_std": 0.47514161467552185, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 978.0} {"timestamp_utc": "2026-04-11T21:29:20Z", "mode": "train", "global_step": 979, "epoch": 0.037805066419524254, "loss": 0.0445, "grad_norm": 2.4577910900115967, "learning_rate": 7.036363636363637e-06, "num_tokens": 2105307.0, "completions/mean_length": 90.25, "completions/min_length": 81.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.25, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.909915566444397, "rewards/meter/std": 0.11405379325151443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.909915566444397, "rewards/total_composite/std": 0.11405379325151443, "reward": 0.909915566444397, "reward_std": 0.11405378580093384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02070813812315464, "sampling/sampling_logp_difference/max": 1.285738229751587, "sampling/importance_sampling_ratio/min": 0.286821186542511, "sampling/importance_sampling_ratio/mean": 1.0027754306793213, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08289938466623425, "clip_ratio/low_mean": 0.007773898891173303, "clip_ratio/low_min": 0.007773898891173303, "clip_ratio/high_mean": 0.01134305214509368, "clip_ratio/high_max": 0.01134305214509368, "clip_ratio/region_mean": 0.019116951036266983, "reward_total_mean": 0.909915566444397, "reward_meter_mean": 0.909915566444397, "reward_meter_std": 0.11405379325151443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.909915566444397, "reward_total_composite_std": 0.11405379325151443, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 979.0} {"timestamp_utc": "2026-04-11T21:29:24Z", "mode": "train", "global_step": 980, "epoch": 0.03784368242199568, "loss": 0.0183, "grad_norm": 1.5766006708145142, "learning_rate": 7.033333333333334e-06, "num_tokens": 2106882.0, "completions/mean_length": 41.875, "completions/min_length": 41.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.06406420469284058, "rewards/meter/std": 0.01977705955505371, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.06406420469284058, "rewards/total_composite/std": 0.01977705955505371, "reward": 0.06406420469284058, "reward_std": 0.01977705769240856, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014269908890128136, "sampling/sampling_logp_difference/max": 2.2084240913391113, "sampling/importance_sampling_ratio/min": 0.10987366735935211, "sampling/importance_sampling_ratio/mean": 0.9980408549308777, "sampling/importance_sampling_ratio/max": 1.1128835678100586, "entropy": 0.04218163015320897, "clip_ratio/low_mean": 0.0029761905316263437, "clip_ratio/low_min": 0.0029761905316263437, "clip_ratio/high_mean": 0.009146341122686863, "clip_ratio/high_max": 0.009146341122686863, "clip_ratio/region_mean": 0.012122531654313207, "reward_total_mean": 0.06406420469284058, "reward_meter_mean": 0.06406420469284058, "reward_meter_std": 0.01977705955505371, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.06406420469284058, "reward_total_composite_std": 0.01977705955505371, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 980.0} {"timestamp_utc": "2026-04-11T21:29:29Z", "mode": "train", "global_step": 981, "epoch": 0.0378822984244671, "loss": 0.0027, "grad_norm": 2.015160083770752, "learning_rate": 7.030303030303031e-06, "num_tokens": 2108769.0, "completions/mean_length": 59.875, "completions/min_length": 59.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.02317265048623085, "rewards/meter/std": 0.007055968511849642, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.02317265048623085, "rewards/total_composite/std": 0.007055968511849642, "reward": 0.02317265048623085, "reward_std": 0.007055968511849642, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00999899860471487, "sampling/sampling_logp_difference/max": 0.9755020141601562, "sampling/importance_sampling_ratio/min": 0.3770030438899994, "sampling/importance_sampling_ratio/mean": 1.0034717321395874, "sampling/importance_sampling_ratio/max": 1.284210205078125, "entropy": 0.04927215212956071, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.02317265048623085, "reward_meter_mean": 0.02317265048623085, "reward_meter_std": 0.007055968511849642, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.02317265048623085, "reward_total_composite_std": 0.007055968511849642, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 981.0} {"timestamp_utc": "2026-04-11T21:29:35Z", "mode": "train", "global_step": 982, "epoch": 0.037920914426938526, "loss": 0.0013, "grad_norm": 2.9196105003356934, "learning_rate": 7.027272727272728e-06, "num_tokens": 2111481.0, "completions/mean_length": 141.0, "completions/min_length": 141.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.0, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.014102931134402752, "rewards/meter/std": 0.007124754600226879, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.011752442456781864, "rewards/total_composite/std": 0.005937295500189066, "reward": 0.011752442456781864, "reward_std": 0.005937295965850353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004966363310813904, "sampling/sampling_logp_difference/max": 1.2700107097625732, "sampling/importance_sampling_ratio/min": 0.2808286249637604, "sampling/importance_sampling_ratio/mean": 1.0011169910430908, "sampling/importance_sampling_ratio/max": 1.434732437133789, "entropy": 0.021313404897227883, "clip_ratio/low_mean": 0.0008865247946232557, "clip_ratio/low_min": 0.0008865247946232557, "clip_ratio/high_mean": 0.002659574383869767, "clip_ratio/high_max": 0.002659574383869767, "clip_ratio/region_mean": 0.003546099178493023, "reward_total_mean": 0.011752442456781864, "reward_meter_mean": 0.014102931134402752, "reward_meter_std": 0.007124754600226879, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.011752442456781864, "reward_total_composite_std": 0.005937295500189066, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 982.0} {"timestamp_utc": "2026-04-11T21:29:41Z", "mode": "train", "global_step": 983, "epoch": 0.03795953042940995, "loss": 0.2112, "grad_norm": 6.859989643096924, "learning_rate": 7.024242424242424e-06, "num_tokens": 2113870.0, "completions/mean_length": 127.625, "completions/min_length": 99.0, "completions/max_length": 198.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.625, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 198.0, "rewards/meter/mean": 0.7576960325241089, "rewards/meter/std": 0.40692663192749023, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.6518734693527222, "rewards/total_composite/std": 0.39248353242874146, "reward": 0.6518734693527222, "reward_std": 0.39248353242874146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04114376753568649, "sampling/sampling_logp_difference/max": 3.9842896461486816, "sampling/importance_sampling_ratio/min": 0.018605656921863556, "sampling/importance_sampling_ratio/mean": 1.0060672760009766, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6028903913684189, "clip_ratio/low_mean": 0.002659574383869767, "clip_ratio/low_min": 0.002659574383869767, "clip_ratio/high_mean": 0.01208286895416677, "clip_ratio/high_max": 0.01208286895416677, "clip_ratio/region_mean": 0.014742443338036537, "reward_total_mean": 0.6518734693527222, "reward_meter_mean": 0.7576960325241089, "reward_meter_std": 0.40692663192749023, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.6518734693527222, "reward_total_composite_std": 0.39248353242874146, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 983.0} {"timestamp_utc": "2026-04-11T21:29:47Z", "mode": "train", "global_step": 984, "epoch": 0.037998146431881374, "loss": -0.0548, "grad_norm": 2.202801465988159, "learning_rate": 7.021212121212122e-06, "num_tokens": 2116276.0, "completions/mean_length": 113.75, "completions/min_length": 103.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.75, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.3314540982246399, "rewards/meter/std": 0.28423401713371277, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3165196180343628, "rewards/total_composite/std": 0.27852126955986023, "reward": 0.3165196180343628, "reward_std": 0.27852123975753784, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01995379850268364, "sampling/sampling_logp_difference/max": 7.068086624145508, "sampling/importance_sampling_ratio/min": 0.0008518615504726768, "sampling/importance_sampling_ratio/mean": 0.9998293519020081, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.042425311636179686, "clip_ratio/low_mean": 0.004831030732020736, "clip_ratio/low_min": 0.004831030732020736, "clip_ratio/high_mean": 0.004976318683475256, "clip_ratio/high_max": 0.004976318683475256, "clip_ratio/region_mean": 0.009807349415495992, "reward_total_mean": 0.3165196180343628, "reward_meter_mean": 0.3314540982246399, "reward_meter_std": 0.28423401713371277, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3165196180343628, "reward_total_composite_std": 0.27852126955986023, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 984.0} {"timestamp_utc": "2026-04-11T21:29:57Z", "mode": "train", "global_step": 985, "epoch": 0.0380367624343528, "loss": -0.0409, "grad_norm": 1.9746062755584717, "learning_rate": 7.018181818181818e-06, "num_tokens": 2119129.0, "completions/mean_length": 284.625, "completions/min_length": 161.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 208.83334350585938, "completions/min_terminated_length": 161.0, "completions/max_terminated_length": 407.0, "rewards/meter/mean": 0.40877169370651245, "rewards/meter/std": 0.22077198326587677, "rewards/count_adherence/mean": 0.675000011920929, "rewards/count_adherence/std": 0.23754701018333435, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/total_composite/mean": 0.14053986966609955, "rewards/total_composite/std": 0.1424238383769989, "reward": 0.14053986966609955, "reward_std": 0.1424238383769989, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05010702461004257, "sampling/sampling_logp_difference/max": 2.6112289428710938, "sampling/importance_sampling_ratio/min": 0.07344422489404678, "sampling/importance_sampling_ratio/mean": 1.0073325634002686, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8109055985696614, "clip_ratio/low_mean": 0.006338990060612559, "clip_ratio/low_min": 0.006338990060612559, "clip_ratio/high_mean": 0.005893339344765991, "clip_ratio/high_max": 0.005893339344765991, "clip_ratio/region_mean": 0.01223232940537855, "reward_total_mean": 0.14053986966609955, "reward_meter_mean": 0.40877169370651245, "reward_meter_std": 0.22077198326587677, "reward_count_adherence_mean": 0.675000011920929, "reward_count_adherence_std": 0.23754701018333435, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_total_composite_mean": 0.14053986966609955, "reward_total_composite_std": 0.1424238383769989, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 985.0} {"timestamp_utc": "2026-04-11T21:30:02Z", "mode": "train", "global_step": 986, "epoch": 0.03807537843682422, "loss": -0.0008, "grad_norm": 2.780010461807251, "learning_rate": 7.015151515151516e-06, "num_tokens": 2120849.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9984778165817261, "rewards/meter/std": 0.00015775838983245194, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984778165817261, "rewards/total_composite/std": 0.00015775838983245194, "reward": 0.9984778165817261, "reward_std": 0.00015774604980833828, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0029492343310266733, "sampling/sampling_logp_difference/max": 0.550690770149231, "sampling/importance_sampling_ratio/min": 0.5765514373779297, "sampling/importance_sampling_ratio/mean": 1.0003331899642944, "sampling/importance_sampling_ratio/max": 1.2361135482788086, "entropy": 0.01241489767562598, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.004098360426723957, "reward_total_mean": 0.9984778165817261, "reward_meter_mean": 0.9984778165817261, "reward_meter_std": 0.00015775838983245194, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984778165817261, "reward_total_composite_std": 0.00015775838983245194, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 986.0} {"timestamp_utc": "2026-04-11T21:30:07Z", "mode": "train", "global_step": 987, "epoch": 0.03811399443929565, "loss": 0.0003, "grad_norm": 9.820355415344238, "learning_rate": 7.0121212121212126e-06, "num_tokens": 2122902.0, "completions/mean_length": 76.625, "completions/min_length": 74.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.625, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.910764217376709, "rewards/meter/std": 0.14799392223358154, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8880656957626343, "rewards/total_composite/std": 0.21216244995594025, "reward": 0.8880656957626343, "reward_std": 0.21216242015361786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014175930060446262, "sampling/sampling_logp_difference/max": 0.7185642123222351, "sampling/importance_sampling_ratio/min": 0.4874516427516937, "sampling/importance_sampling_ratio/mean": 1.0007253885269165, "sampling/importance_sampling_ratio/max": 1.5090887546539307, "entropy": 0.0668493090197444, "clip_ratio/low_mean": 0.0016447368543595076, "clip_ratio/low_min": 0.0016447368543595076, "clip_ratio/high_mean": 0.014278275077231228, "clip_ratio/high_max": 0.014278275077231228, "clip_ratio/region_mean": 0.015923011931590736, "reward_total_mean": 0.8880656957626343, "reward_meter_mean": 0.910764217376709, "reward_meter_std": 0.14799392223358154, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8880656957626343, "reward_total_composite_std": 0.21216244995594025, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 987.0} {"timestamp_utc": "2026-04-11T21:30:12Z", "mode": "train", "global_step": 988, "epoch": 0.03815261044176707, "loss": 0.2499, "grad_norm": 9.86164379119873, "learning_rate": 7.00909090909091e-06, "num_tokens": 2124400.0, "completions/mean_length": 30.25, "completions/min_length": 26.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.25, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9256762266159058, "rewards/meter/std": 0.011381997726857662, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8066024780273438, "rewards/total_composite/std": 0.32593393325805664, "reward": 0.8066024780273438, "reward_std": 0.32593396306037903, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02972019836306572, "sampling/sampling_logp_difference/max": 2.015109062194824, "sampling/importance_sampling_ratio/min": 0.1333058625459671, "sampling/importance_sampling_ratio/mean": 0.9983457922935486, "sampling/importance_sampling_ratio/max": 1.3473082780838013, "entropy": 0.11152052227407694, "clip_ratio/low_mean": 0.0024509804788976908, "clip_ratio/low_min": 0.0024509804788976908, "clip_ratio/high_mean": 0.03225478297099471, "clip_ratio/high_max": 0.03225478297099471, "clip_ratio/region_mean": 0.0347057634498924, "reward_total_mean": 0.8066024780273438, "reward_meter_mean": 0.9256762266159058, "reward_meter_std": 0.011381997726857662, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8066024780273438, "reward_total_composite_std": 0.32593393325805664, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 988.0} {"timestamp_utc": "2026-04-11T21:30:18Z", "mode": "train", "global_step": 989, "epoch": 0.038191226444238495, "loss": -0.0064, "grad_norm": 3.7689943313598633, "learning_rate": 7.006060606060606e-06, "num_tokens": 2126530.0, "completions/mean_length": 88.25, "completions/min_length": 85.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.25, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.2954156994819641, "rewards/meter/std": 0.2895492911338806, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.2954156994819641, "rewards/total_composite/std": 0.2895492911338806, "reward": 0.2954156994819641, "reward_std": 0.289549320936203, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02267267368733883, "sampling/sampling_logp_difference/max": 0.9028096199035645, "sampling/importance_sampling_ratio/min": 0.405428946018219, "sampling/importance_sampling_ratio/mean": 1.0014500617980957, "sampling/importance_sampling_ratio/max": 1.8068927526474, "entropy": 0.12002238910645247, "clip_ratio/low_mean": 0.008599534747190773, "clip_ratio/low_min": 0.008599534747190773, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.008599534747190773, "reward_total_mean": 0.2954156994819641, "reward_meter_mean": 0.2954156994819641, "reward_meter_std": 0.2895492911338806, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.2954156994819641, "reward_total_composite_std": 0.2895492911338806, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 989.0} {"timestamp_utc": "2026-04-11T21:30:26Z", "mode": "train", "global_step": 990, "epoch": 0.03822984244670992, "loss": -0.0316, "grad_norm": 2.919459581375122, "learning_rate": 7.0030303030303035e-06, "num_tokens": 2130557.0, "completions/mean_length": 269.375, "completions/min_length": 244.0, "completions/max_length": 305.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 269.375, "completions/min_terminated_length": 244.0, "completions/max_terminated_length": 305.0, "rewards/meter/mean": 0.725569486618042, "rewards/meter/std": 0.37557560205459595, "rewards/count_adherence/mean": 0.8392857313156128, "rewards/count_adherence/std": 0.05050762742757797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6042459011077881, "rewards/total_composite/std": 0.3114357888698578, "reward": 0.6042459011077881, "reward_std": 0.3114357888698578, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006369084119796753, "sampling/sampling_logp_difference/max": 1.6207554340362549, "sampling/importance_sampling_ratio/min": 0.19774925708770752, "sampling/importance_sampling_ratio/mean": 0.9994811415672302, "sampling/importance_sampling_ratio/max": 1.661027431488037, "entropy": 0.02545841468963772, "clip_ratio/low_mean": 0.0020011526066809893, "clip_ratio/low_min": 0.0020011526066809893, "clip_ratio/high_mean": 0.0017354230512864888, "clip_ratio/high_max": 0.0017354230512864888, "clip_ratio/region_mean": 0.003736575657967478, "reward_total_mean": 0.6042459011077881, "reward_meter_mean": 0.725569486618042, "reward_meter_std": 0.37557560205459595, "reward_count_adherence_mean": 0.8392857313156128, "reward_count_adherence_std": 0.05050762742757797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6042459011077881, "reward_total_composite_std": 0.3114357888698578, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 990.0} {"timestamp_utc": "2026-04-11T21:30:31Z", "mode": "train", "global_step": 991, "epoch": 0.03826845844918134, "loss": -0.0175, "grad_norm": 3.4146828651428223, "learning_rate": 7e-06, "num_tokens": 2132616.0, "completions/mean_length": 93.375, "completions/min_length": 88.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.375, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9935424327850342, "rewards/meter/std": 0.001386665622703731, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9935424327850342, "rewards/total_composite/std": 0.001386665622703731, "reward": 0.9935424327850342, "reward_std": 0.0013866751687601209, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026832029223442078, "sampling/sampling_logp_difference/max": 5.278292179107666, "sampling/importance_sampling_ratio/min": 0.00510113500058651, "sampling/importance_sampling_ratio/mean": 1.0043511390686035, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0766637921333313, "clip_ratio/low_mean": 0.010829056613147259, "clip_ratio/low_min": 0.010829056613147259, "clip_ratio/high_mean": 0.0012886597542092204, "clip_ratio/high_max": 0.0012886597542092204, "clip_ratio/region_mean": 0.01211771636735648, "reward_total_mean": 0.9935424327850342, "reward_meter_mean": 0.9935424327850342, "reward_meter_std": 0.001386665622703731, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9935424327850342, "reward_total_composite_std": 0.001386665622703731, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 991.0} {"timestamp_utc": "2026-04-11T21:30:38Z", "mode": "train", "global_step": 992, "epoch": 0.03830707445165277, "loss": 0.0159, "grad_norm": 2.6965482234954834, "learning_rate": 6.996969696969698e-06, "num_tokens": 2135377.0, "completions/mean_length": 157.125, "completions/min_length": 151.0, "completions/max_length": 169.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.125, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 169.0, "rewards/meter/mean": 0.989780068397522, "rewards/meter/std": 0.0013144243275746703, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7423350811004639, "rewards/total_composite/std": 0.000985805643722415, "reward": 0.7423350811004639, "reward_std": 0.0009857891127467155, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009137184359133244, "sampling/sampling_logp_difference/max": 2.072084426879883, "sampling/importance_sampling_ratio/min": 0.1259230375289917, "sampling/importance_sampling_ratio/mean": 1.0015301704406738, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02171965758316219, "clip_ratio/low_mean": 0.002329192589968443, "clip_ratio/low_min": 0.002329192589968443, "clip_ratio/high_mean": 0.0056129166623577476, "clip_ratio/high_max": 0.0056129166623577476, "clip_ratio/region_mean": 0.00794210925232619, "reward_total_mean": 0.7423350811004639, "reward_meter_mean": 0.989780068397522, "reward_meter_std": 0.0013144243275746703, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7423350811004639, "reward_total_composite_std": 0.000985805643722415, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 992.0} {"timestamp_utc": "2026-04-11T21:30:43Z", "mode": "train", "global_step": 993, "epoch": 0.03834569045412419, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.993939393939394e-06, "num_tokens": 2136841.0, "completions/mean_length": 31.0, "completions/min_length": 31.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.001403667964041233, "sampling/sampling_logp_difference/max": 0.041133493185043335, "sampling/importance_sampling_ratio/min": 0.9597010016441345, "sampling/importance_sampling_ratio/mean": 1.0010855197906494, "sampling/importance_sampling_ratio/max": 1.0404589176177979, "entropy": 0.0084037822089158, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 993.0} {"timestamp_utc": "2026-04-11T21:30:48Z", "mode": "train", "global_step": 994, "epoch": 0.038384306456595615, "loss": 0.0161, "grad_norm": 5.34966516494751, "learning_rate": 6.990909090909092e-06, "num_tokens": 2138634.0, "completions/mean_length": 53.125, "completions/min_length": 50.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8893836736679077, "rewards/meter/std": 0.04377932846546173, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8893836736679077, "rewards/total_composite/std": 0.04377932846546173, "reward": 0.8893836736679077, "reward_std": 0.04377933219075203, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01280057430267334, "sampling/sampling_logp_difference/max": 0.604201078414917, "sampling/importance_sampling_ratio/min": 0.5465108752250671, "sampling/importance_sampling_ratio/mean": 1.0018515586853027, "sampling/importance_sampling_ratio/max": 1.511669635772705, "entropy": 0.0857058372348547, "clip_ratio/low_mean": 0.0069460326340049505, "clip_ratio/low_min": 0.0069460326340049505, "clip_ratio/high_mean": 0.004999999888241291, "clip_ratio/high_max": 0.004999999888241291, "clip_ratio/region_mean": 0.011946032522246242, "reward_total_mean": 0.8893836736679077, "reward_meter_mean": 0.8893836736679077, "reward_meter_std": 0.04377932846546173, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8893836736679077, "reward_total_composite_std": 0.04377932846546173, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 994.0} {"timestamp_utc": "2026-04-11T21:30:58Z", "mode": "train", "global_step": 995, "epoch": 0.03842292245906704, "loss": -0.1685, "grad_norm": 1.5435454845428467, "learning_rate": 6.987878787878788e-06, "num_tokens": 2142040.0, "completions/mean_length": 250.75, "completions/min_length": 169.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 213.42857360839844, "completions/min_terminated_length": 169.0, "completions/max_terminated_length": 251.0, "rewards/meter/mean": 0.40631982684135437, "rewards/meter/std": 0.45718684792518616, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.3808841407299042, "rewards/total_composite/std": 0.4337118864059448, "reward": 0.3808841407299042, "reward_std": 0.4337119162082672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01201361883431673, "sampling/sampling_logp_difference/max": 4.515697956085205, "sampling/importance_sampling_ratio/min": 0.010935969650745392, "sampling/importance_sampling_ratio/mean": 0.9998841881752014, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.031067739007994533, "clip_ratio/low_mean": 0.003973921644501388, "clip_ratio/low_min": 0.003973921644501388, "clip_ratio/high_mean": 0.001066165801603347, "clip_ratio/high_max": 0.001066165801603347, "clip_ratio/region_mean": 0.005040087446104735, "reward_total_mean": 0.3808841407299042, "reward_meter_mean": 0.40631982684135437, "reward_meter_std": 0.45718684792518616, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.3808841407299042, "reward_total_composite_std": 0.4337118864059448, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 995.0} {"timestamp_utc": "2026-04-11T21:31:04Z", "mode": "train", "global_step": 996, "epoch": 0.038461538461538464, "loss": -0.0218, "grad_norm": 3.508453130722046, "learning_rate": 6.984848484848485e-06, "num_tokens": 2144126.0, "completions/mean_length": 88.75, "completions/min_length": 85.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.75, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9983739852905273, "rewards/meter/std": 0.0003010180371347815, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9983739852905273, "rewards/total_composite/std": 0.0003010180371347815, "reward": 0.9983739852905273, "reward_std": 0.0003010221407748759, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0038979947566986084, "sampling/sampling_logp_difference/max": 1.1575075387954712, "sampling/importance_sampling_ratio/min": 0.3142684996128082, "sampling/importance_sampling_ratio/mean": 0.9988738894462585, "sampling/importance_sampling_ratio/max": 1.1453224420547485, "entropy": 0.00855121889617294, "clip_ratio/low_mean": 0.00147058826405555, "clip_ratio/low_min": 0.00147058826405555, "clip_ratio/high_mean": 0.0013736264081671834, "clip_ratio/high_max": 0.0013736264081671834, "clip_ratio/region_mean": 0.0028442146722227335, "reward_total_mean": 0.9983739852905273, "reward_meter_mean": 0.9983739852905273, "reward_meter_std": 0.0003010180371347815, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9983739852905273, "reward_total_composite_std": 0.0003010180371347815, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 996.0} {"timestamp_utc": "2026-04-11T21:31:09Z", "mode": "train", "global_step": 997, "epoch": 0.03850015446400989, "loss": 0.0184, "grad_norm": 4.285051345825195, "learning_rate": 6.981818181818183e-06, "num_tokens": 2145922.0, "completions/mean_length": 61.5, "completions/min_length": 61.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9900733232498169, "rewards/meter/std": 0.0007385181379504502, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9900733232498169, "rewards/total_composite/std": 0.0007385181379504502, "reward": 0.9900733232498169, "reward_std": 0.0007385211647488177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0019282199209555984, "sampling/sampling_logp_difference/max": 0.14210748672485352, "sampling/importance_sampling_ratio/min": 0.8675280213356018, "sampling/importance_sampling_ratio/mean": 1.001381278038025, "sampling/importance_sampling_ratio/max": 1.0559881925582886, "entropy": 0.013517249142751098, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9900733232498169, "reward_meter_mean": 0.9900733232498169, "reward_meter_std": 0.0007385181379504502, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9900733232498169, "reward_total_composite_std": 0.0007385181379504502, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 997.0} {"timestamp_utc": "2026-04-11T21:31:17Z", "mode": "train", "global_step": 998, "epoch": 0.03853877046648131, "loss": 0.0261, "grad_norm": 1.264242172241211, "learning_rate": 6.978787878787879e-06, "num_tokens": 2149920.0, "completions/mean_length": 312.75, "completions/min_length": 294.0, "completions/max_length": 326.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 312.75, "completions/min_terminated_length": 294.0, "completions/max_terminated_length": 326.0, "rewards/meter/mean": 0.008165713399648666, "rewards/meter/std": 0.010855883359909058, "rewards/count_adherence/mean": 0.9666666984558105, "rewards/count_adherence/std": 0.035634830594062805, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.007689647376537323, "rewards/total_composite/std": 0.010093861259520054, "reward": 0.007689647376537323, "reward_std": 0.010093861259520054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009022814221680164, "sampling/sampling_logp_difference/max": 1.5105669498443604, "sampling/importance_sampling_ratio/min": 0.2207847684621811, "sampling/importance_sampling_ratio/mean": 1.000508427619934, "sampling/importance_sampling_ratio/max": 1.566821575164795, "entropy": 0.03884409740567207, "clip_ratio/low_mean": 0.001570199674461037, "clip_ratio/low_min": 0.001570199674461037, "clip_ratio/high_mean": 0.0025131339207291603, "clip_ratio/high_max": 0.0025131339207291603, "clip_ratio/region_mean": 0.004083333595190197, "reward_total_mean": 0.007689647376537323, "reward_meter_mean": 0.008165713399648666, "reward_meter_std": 0.010855883359909058, "reward_count_adherence_mean": 0.9666666984558105, "reward_count_adherence_std": 0.035634830594062805, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.007689647376537323, "reward_total_composite_std": 0.010093861259520054, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 998.0} {"timestamp_utc": "2026-04-11T21:31:22Z", "mode": "train", "global_step": 999, "epoch": 0.038577386468952736, "loss": -0.0029, "grad_norm": 15.84067440032959, "learning_rate": 6.975757575757577e-06, "num_tokens": 2151534.0, "completions/mean_length": 37.75, "completions/min_length": 36.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.3695641756057739, "rewards/meter/std": 0.36389413475990295, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3695641756057739, "rewards/total_composite/std": 0.36389413475990295, "reward": 0.3695641756057739, "reward_std": 0.36389410495758057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0329262800514698, "sampling/sampling_logp_difference/max": 1.2456717491149902, "sampling/importance_sampling_ratio/min": 0.2877475619316101, "sampling/importance_sampling_ratio/mean": 1.005554437637329, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.086491241119802, "clip_ratio/low_mean": 0.02047239593230188, "clip_ratio/low_min": 0.02047239593230188, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.02047239593230188, "reward_total_mean": 0.3695641756057739, "reward_meter_mean": 0.3695641756057739, "reward_meter_std": 0.36389413475990295, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3695641756057739, "reward_total_composite_std": 0.36389413475990295, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 999.0} {"timestamp_utc": "2026-04-11T21:31:27Z", "mode": "train", "global_step": 1000, "epoch": 0.03861600247142416, "loss": 0.0253, "grad_norm": 5.247101306915283, "learning_rate": 6.9727272727272735e-06, "num_tokens": 2153476.0, "completions/mean_length": 89.75, "completions/min_length": 85.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.75, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.7209616899490356, "rewards/meter/std": 0.11303994804620743, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7209616899490356, "rewards/total_composite/std": 0.11303994804620743, "reward": 0.7209616899490356, "reward_std": 0.11303995549678802, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020473238080739975, "sampling/sampling_logp_difference/max": 2.644613027572632, "sampling/importance_sampling_ratio/min": 0.07103283703327179, "sampling/importance_sampling_ratio/mean": 1.0050361156463623, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05528395110741258, "clip_ratio/low_mean": 0.004979395656846464, "clip_ratio/low_min": 0.004979395656846464, "clip_ratio/high_mean": 0.0014534883666783571, "clip_ratio/high_max": 0.0014534883666783571, "clip_ratio/region_mean": 0.006432884023524821, "reward_total_mean": 0.7209616899490356, "reward_meter_mean": 0.7209616899490356, "reward_meter_std": 0.11303994804620743, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7209616899490356, "reward_total_composite_std": 0.11303994804620743, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1000.0} {"timestamp_utc": "2026-04-11T21:32:49Z", "mode": "eval", "global_step": 1000, "epoch": 0.03861600247142416, "eval_loss": NaN, "eval_runtime": 81.6951, "eval_samples_per_second": 1.273, "eval_steps_per_second": 0.159, "eval_num_tokens": 2153476.0, "eval_completions/mean_length": 211.65384615384616, "eval_completions/min_length": 50.0, "eval_completions/max_length": 436.3076923076923, "eval_completions/clipped_ratio": 0.04807692307692308, "eval_completions/mean_terminated_length": 196.46566420335037, "eval_completions/min_terminated_length": 50.0, "eval_completions/max_terminated_length": 375.2307692307692, "eval_rewards/meter/mean": 0.6936025619506836, "eval_rewards/meter/std": 0.4170892101067763, "eval_rewards/count_adherence/mean": 0.9188327697607187, "eval_rewards/count_adherence/std": 0.0954194341141444, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/total_composite/mean": 0.6325189058597271, "eval_rewards/total_composite/std": 0.39416128396987915, "eval_reward": 0.6325189058597271, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.00360978885482137, "eval_sampling/sampling_logp_difference/max": 0.699675963475154, "eval_sampling/importance_sampling_ratio/min": 0.5284584256318899, "eval_sampling/importance_sampling_ratio/mean": 1.0007417064446669, "eval_sampling/importance_sampling_ratio/max": 1.2181079571063702, "eval_entropy": 0.027736100296561535, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6325189058597271, "eval_reward_meter_mean": 0.6936025619506836, "eval_reward_meter_std": 0.4170892101067763, "eval_reward_count_adherence_mean": 0.9188327697607187, "eval_reward_count_adherence_std": 0.0954194341141444, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_total_composite_mean": 0.6325189058597271, "eval_reward_total_composite_std": 0.39416128396987915, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1000.0} {"timestamp_utc": "2026-04-11T21:32:57Z", "mode": "train", "global_step": 1001, "epoch": 0.038654618473895584, "loss": 0.0022, "grad_norm": 3.507573127746582, "learning_rate": 6.969696969696971e-06, "num_tokens": 2155212.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8180665969848633, "rewards/meter/std": 0.005837064702063799, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8180665969848633, "rewards/total_composite/std": 0.005837064702063799, "reward": 0.8180665969848633, "reward_std": 0.005837073549628258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004756717476993799, "sampling/sampling_logp_difference/max": 1.066084623336792, "sampling/importance_sampling_ratio/min": 0.9404751062393188, "sampling/importance_sampling_ratio/mean": 1.003198504447937, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.024478797102347016, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004098360426723957, "reward_total_mean": 0.8180665969848633, "reward_meter_mean": 0.8180665969848633, "reward_meter_std": 0.005837064702063799, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8180665969848633, "reward_total_composite_std": 0.005837064702063799, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1001.0} {"timestamp_utc": "2026-04-11T21:33:07Z", "mode": "train", "global_step": 1002, "epoch": 0.03869323447636701, "loss": -0.2057, "grad_norm": 1.5992178916931152, "learning_rate": 6.966666666666667e-06, "num_tokens": 2157336.0, "completions/mean_length": 155.5, "completions/min_length": 91.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 104.5714340209961, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9517431855201721, "rewards/meter/std": 0.12741346657276154, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8306448459625244, "rewards/total_composite/std": 0.3551715612411499, "reward": 0.8306448459625244, "reward_std": 0.3551715314388275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008776551112532616, "sampling/sampling_logp_difference/max": 1.5235629081726074, "sampling/importance_sampling_ratio/min": 0.2179340273141861, "sampling/importance_sampling_ratio/mean": 1.0019506216049194, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02280411496758461, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004720762372016907, "clip_ratio/high_max": 0.004720762372016907, "clip_ratio/region_mean": 0.004720762372016907, "reward_total_mean": 0.8306448459625244, "reward_meter_mean": 0.9517431855201721, "reward_meter_std": 0.12741346657276154, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8306448459625244, "reward_total_composite_std": 0.3551715612411499, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1002.0} {"timestamp_utc": "2026-04-11T21:33:11Z", "mode": "train", "global_step": 1003, "epoch": 0.03873185047883843, "loss": 0.0053, "grad_norm": 2.10730242729187, "learning_rate": 6.963636363636364e-06, "num_tokens": 2158952.0, "completions/mean_length": 38.0, "completions/min_length": 38.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8173251152038574, "rewards/meter/std": 0.028324369341135025, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8173251152038574, "rewards/total_composite/std": 0.028324369341135025, "reward": 0.8173251152038574, "reward_std": 0.028324367478489876, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008000357076525688, "sampling/sampling_logp_difference/max": 0.8077940940856934, "sampling/importance_sampling_ratio/min": 0.4458405077457428, "sampling/importance_sampling_ratio/mean": 0.9973918795585632, "sampling/importance_sampling_ratio/max": 1.0620988607406616, "entropy": 0.033309838734567165, "clip_ratio/low_mean": 0.00657894741743803, "clip_ratio/low_min": 0.00657894741743803, "clip_ratio/high_mean": 0.003289473708719015, "clip_ratio/high_max": 0.003289473708719015, "clip_ratio/region_mean": 0.009868421126157045, "reward_total_mean": 0.8173251152038574, "reward_meter_mean": 0.8173251152038574, "reward_meter_std": 0.028324369341135025, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8173251152038574, "reward_total_composite_std": 0.028324369341135025, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1003.0} {"timestamp_utc": "2026-04-11T21:33:17Z", "mode": "train", "global_step": 1004, "epoch": 0.038770466481309857, "loss": -0.0005, "grad_norm": 3.0838623046875, "learning_rate": 6.960606060606061e-06, "num_tokens": 2160752.0, "completions/mean_length": 65.0, "completions/min_length": 65.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9935441017150879, "rewards/meter/std": 0.002191033447161317, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9935441017150879, "rewards/total_composite/std": 0.002191033447161317, "reward": 0.9935441017150879, "reward_std": 0.0021910332143306732, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005409308709204197, "sampling/sampling_logp_difference/max": 0.29759836196899414, "sampling/importance_sampling_ratio/min": 0.7425995469093323, "sampling/importance_sampling_ratio/mean": 1.0010415315628052, "sampling/importance_sampling_ratio/max": 1.2240910530090332, "entropy": 0.04200784815475345, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/high_mean": 0.005769230774603784, "clip_ratio/high_max": 0.005769230774603784, "clip_ratio/region_mean": 0.007692307699471712, "reward_total_mean": 0.9935441017150879, "reward_meter_mean": 0.9935441017150879, "reward_meter_std": 0.002191033447161317, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9935441017150879, "reward_total_composite_std": 0.002191033447161317, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1004.0} {"timestamp_utc": "2026-04-11T21:33:22Z", "mode": "train", "global_step": 1005, "epoch": 0.03880908248378128, "loss": -0.0056, "grad_norm": 5.190495014190674, "learning_rate": 6.957575757575759e-06, "num_tokens": 2162985.0, "completions/mean_length": 111.125, "completions/min_length": 97.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.125, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.05873946100473404, "rewards/meter/std": 0.0528215616941452, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.04519527032971382, "rewards/total_composite/std": 0.038662139326334, "reward": 0.04519527032971382, "reward_std": 0.0386621356010437, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013827898539602757, "sampling/sampling_logp_difference/max": 0.8459677696228027, "sampling/importance_sampling_ratio/min": 0.4291418492794037, "sampling/importance_sampling_ratio/mean": 1.001181721687317, "sampling/importance_sampling_ratio/max": 1.5605558156967163, "entropy": 0.08433546568267047, "clip_ratio/low_mean": 0.0048094624653458595, "clip_ratio/low_min": 0.0048094624653458595, "clip_ratio/high_mean": 0.00327124388422817, "clip_ratio/high_max": 0.00327124388422817, "clip_ratio/region_mean": 0.00808070634957403, "reward_total_mean": 0.04519527032971382, "reward_meter_mean": 0.05873946100473404, "reward_meter_std": 0.0528215616941452, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.04519527032971382, "reward_total_composite_std": 0.038662139326334, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1005.0} {"timestamp_utc": "2026-04-11T21:33:27Z", "mode": "train", "global_step": 1006, "epoch": 0.038847698486252705, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.954545454545455e-06, "num_tokens": 2164713.0, "completions/mean_length": 46.0, "completions/min_length": 46.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.14252464473247528, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.14252464473247528, "rewards/total_composite/std": 0.0, "reward": 0.14252464473247528, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0049051851965487, "sampling/sampling_logp_difference/max": 0.07062534987926483, "sampling/importance_sampling_ratio/min": 0.953210711479187, "sampling/importance_sampling_ratio/mean": 1.0042462348937988, "sampling/importance_sampling_ratio/max": 1.0731791257858276, "entropy": 0.04377919761464, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.14252464473247528, "reward_meter_mean": 0.14252464473247528, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.14252464473247528, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1006.0} {"timestamp_utc": "2026-04-11T21:33:34Z", "mode": "train", "global_step": 1007, "epoch": 0.03888631448872413, "loss": -0.0419, "grad_norm": 2.1360721588134766, "learning_rate": 6.951515151515153e-06, "num_tokens": 2167798.0, "completions/mean_length": 207.625, "completions/min_length": 194.0, "completions/max_length": 242.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 207.625, "completions/min_terminated_length": 194.0, "completions/max_terminated_length": 242.0, "rewards/meter/mean": 0.9658156037330627, "rewards/meter/std": 0.011747553013265133, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8278419971466064, "rewards/total_composite/std": 0.010069313459098339, "reward": 0.8278419971466064, "reward_std": 0.010069318115711212, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010072649456560612, "sampling/sampling_logp_difference/max": 1.52454674243927, "sampling/importance_sampling_ratio/min": 0.21771973371505737, "sampling/importance_sampling_ratio/mean": 1.0003001689910889, "sampling/importance_sampling_ratio/max": 1.739989995956421, "entropy": 0.028256691992282867, "clip_ratio/low_mean": 0.005001572601031512, "clip_ratio/low_min": 0.005001572601031512, "clip_ratio/high_mean": 0.0023369172704406083, "clip_ratio/high_max": 0.0023369172704406083, "clip_ratio/region_mean": 0.00733848987147212, "reward_total_mean": 0.8278419971466064, "reward_meter_mean": 0.9658156037330627, "reward_meter_std": 0.011747553013265133, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8278419971466064, "reward_total_composite_std": 0.010069313459098339, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1007.0} {"timestamp_utc": "2026-04-11T21:33:40Z", "mode": "train", "global_step": 1008, "epoch": 0.03892493049119555, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.948484848484849e-06, "num_tokens": 2170518.0, "completions/mean_length": 166.0, "completions/min_length": 166.0, "completions/max_length": 166.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.0, "completions/min_terminated_length": 166.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7489440441131592, "rewards/total_composite/std": 0.0, "reward": 0.7489440441131592, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00011090007319580764, "sampling/sampling_logp_difference/max": 0.017774349078536034, "sampling/importance_sampling_ratio/min": 0.9823826551437378, "sampling/importance_sampling_ratio/mean": 1.0000582933425903, "sampling/importance_sampling_ratio/max": 1.0169156789779663, "entropy": 0.0009103744814638048, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7489440441131592, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7489440441131592, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1008.0} {"timestamp_utc": "2026-04-11T21:33:50Z", "mode": "train", "global_step": 1009, "epoch": 0.03896354649366698, "loss": -0.2117, "grad_norm": 0.5083670616149902, "learning_rate": 6.945454545454546e-06, "num_tokens": 2172857.0, "completions/mean_length": 159.375, "completions/min_length": 109.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 109.00000762939453, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.8148025870323181, "rewards/meter/std": 0.27071505784988403, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.5311336517333984, "rewards/total_composite/std": 0.21461041271686554, "reward": 0.5311336517333984, "reward_std": 0.21461039781570435, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0029847025871276855, "sampling/sampling_logp_difference/max": 0.17074882984161377, "sampling/importance_sampling_ratio/min": 0.8430333137512207, "sampling/importance_sampling_ratio/mean": 1.0008081197738647, "sampling/importance_sampling_ratio/max": 1.163157343864441, "entropy": 0.0178952447604388, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0034403668250888586, "clip_ratio/high_max": 0.0034403668250888586, "clip_ratio/region_mean": 0.0034403668250888586, "reward_total_mean": 0.5311336517333984, "reward_meter_mean": 0.8148025870323181, "reward_meter_std": 0.27071505784988403, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.5311336517333984, "reward_total_composite_std": 0.21461041271686554, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1009.0} {"timestamp_utc": "2026-04-11T21:34:02Z", "mode": "train", "global_step": 1010, "epoch": 0.0390021624961384, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.942424242424243e-06, "num_tokens": 2175073.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9637309312820435, "rewards/meter/std": 0.08817882090806961, "rewards/count_adherence/mean": 0.8026316165924072, "rewards/count_adherence/std": 0.073091059923172, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7736469507217407, "rewards/total_composite/std": 0.10139304399490356, "reward": 0.7736469507217407, "reward_std": 0.10139304399490356, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7736469507217407, "reward_meter_mean": 0.9637309312820435, "reward_meter_std": 0.08817882090806961, "reward_count_adherence_mean": 0.8026316165924072, "reward_count_adherence_std": 0.073091059923172, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7736469507217407, "reward_total_composite_std": 0.10139304399490356, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1010.0} {"timestamp_utc": "2026-04-11T21:34:07Z", "mode": "train", "global_step": 1011, "epoch": 0.039040778498609825, "loss": -0.0011, "grad_norm": 3.0494747161865234, "learning_rate": 6.93939393939394e-06, "num_tokens": 2176843.0, "completions/mean_length": 60.25, "completions/min_length": 58.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9968514442443848, "rewards/meter/std": 4.425136648933403e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9968514442443848, "rewards/total_composite/std": 4.425136648933403e-05, "reward": 0.9968514442443848, "reward_std": 4.425585575518198e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007558867335319519, "sampling/sampling_logp_difference/max": 0.6077744960784912, "sampling/importance_sampling_ratio/min": 0.5445614457130432, "sampling/importance_sampling_ratio/mean": 1.0004818439483643, "sampling/importance_sampling_ratio/max": 1.8261241912841797, "entropy": 0.0351905336137861, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/high_mean": 0.006250000325962901, "clip_ratio/high_max": 0.006250000325962901, "clip_ratio/region_mean": 0.00829918053932488, "reward_total_mean": 0.9968514442443848, "reward_meter_mean": 0.9968514442443848, "reward_meter_std": 4.425136648933403e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9968514442443848, "reward_total_composite_std": 4.425136648933403e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1011.0} {"timestamp_utc": "2026-04-11T21:34:12Z", "mode": "train", "global_step": 1012, "epoch": 0.03907939450108125, "loss": 0.0108, "grad_norm": 2.7127881050109863, "learning_rate": 6.936363636363636e-06, "num_tokens": 2178536.0, "completions/mean_length": 51.625, "completions/min_length": 50.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.625, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8671625852584839, "rewards/meter/std": 0.2607937753200531, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8671625852584839, "rewards/total_composite/std": 0.2607937753200531, "reward": 0.8671625852584839, "reward_std": 0.2607937753200531, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01780473254621029, "sampling/sampling_logp_difference/max": 1.3962616920471191, "sampling/importance_sampling_ratio/min": 0.2475205510854721, "sampling/importance_sampling_ratio/mean": 1.0002473592758179, "sampling/importance_sampling_ratio/max": 1.451189398765564, "entropy": 0.06056457059457898, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/high_mean": 0.007358490489423275, "clip_ratio/high_max": 0.007358490489423275, "clip_ratio/region_mean": 0.009673305321484804, "reward_total_mean": 0.8671625852584839, "reward_meter_mean": 0.8671625852584839, "reward_meter_std": 0.2607937753200531, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8671625852584839, "reward_total_composite_std": 0.2607937753200531, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1012.0} {"timestamp_utc": "2026-04-11T21:34:17Z", "mode": "train", "global_step": 1013, "epoch": 0.03911801050355267, "loss": 0.0607, "grad_norm": 2.744403839111328, "learning_rate": 6.9333333333333344e-06, "num_tokens": 2180608.0, "completions/mean_length": 101.0, "completions/min_length": 85.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.0, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.1489562839269638, "rewards/meter/std": 0.3414601683616638, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.1379307359457016, "rewards/total_composite/std": 0.3453916013240814, "reward": 0.1379307359457016, "reward_std": 0.3453916013240814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03784632310271263, "sampling/sampling_logp_difference/max": 10.555170059204102, "sampling/importance_sampling_ratio/min": 2.6058407456730492e-05, "sampling/importance_sampling_ratio/mean": 0.9965633749961853, "sampling/importance_sampling_ratio/max": 1.9995967149734497, "entropy": 0.09084944939240813, "clip_ratio/low_mean": 0.015935942879877985, "clip_ratio/low_min": 0.015935942879877985, "clip_ratio/high_mean": 0.004411764908581972, "clip_ratio/high_max": 0.004411764908581972, "clip_ratio/region_mean": 0.020347707788459957, "reward_total_mean": 0.1379307359457016, "reward_meter_mean": 0.1489562839269638, "reward_meter_std": 0.3414601683616638, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.1379307359457016, "reward_total_composite_std": 0.3453916013240814, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1013.0} {"timestamp_utc": "2026-04-11T21:34:27Z", "mode": "train", "global_step": 1014, "epoch": 0.0391566265060241, "loss": 0.333, "grad_norm": 0.46811041235923767, "learning_rate": 6.930303030303031e-06, "num_tokens": 2184328.0, "completions/mean_length": 400.0, "completions/min_length": 334.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 362.66668701171875, "completions/min_terminated_length": 334.0, "completions/max_terminated_length": 391.0, "rewards/meter/mean": 0.2825581431388855, "rewards/meter/std": 0.4369811415672302, "rewards/count_adherence/mean": 0.3928571343421936, "rewards/count_adherence/std": 0.1266293227672577, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.06818556040525436, "rewards/total_composite/std": 0.09754088521003723, "reward": 0.06818556040525436, "reward_std": 0.09754088521003723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008382507599890232, "sampling/sampling_logp_difference/max": 1.3366272449493408, "sampling/importance_sampling_ratio/min": 0.26273030042648315, "sampling/importance_sampling_ratio/mean": 0.9992491602897644, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.016180144273675978, "clip_ratio/low_mean": 0.00381775782443583, "clip_ratio/low_min": 0.00381775782443583, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.00381775782443583, "reward_total_mean": 0.06818556040525436, "reward_meter_mean": 0.2825581431388855, "reward_meter_std": 0.4369811415672302, "reward_count_adherence_mean": 0.3928571343421936, "reward_count_adherence_std": 0.1266293227672577, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.06818556040525436, "reward_total_composite_std": 0.09754088521003723, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1014.0} {"timestamp_utc": "2026-04-11T21:34:35Z", "mode": "train", "global_step": 1015, "epoch": 0.03919524250849552, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.927272727272728e-06, "num_tokens": 2187804.0, "completions/mean_length": 248.5, "completions/min_length": 241.0, "completions/max_length": 256.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 248.5, "completions/min_terminated_length": 241.0, "completions/max_terminated_length": 256.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.4000000059604645, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3994368314743042, "rewards/total_composite/std": 0.0, "reward": 0.3994368314743042, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.003100335830822587, "sampling/sampling_logp_difference/max": 2.1027605533599854, "sampling/importance_sampling_ratio/min": 0.12211884558200836, "sampling/importance_sampling_ratio/mean": 1.0011197328567505, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.002095947915222496, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.3994368314743042, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.4000000059604645, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3994368314743042, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1015.0} {"timestamp_utc": "2026-04-11T21:34:40Z", "mode": "train", "global_step": 1016, "epoch": 0.039233858510966946, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.9242424242424245e-06, "num_tokens": 2189900.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.49929603934288025, "rewards/total_composite/std": 0.0, "reward": 0.49929603934288025, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00016283347213175148, "sampling/sampling_logp_difference/max": 0.018423333764076233, "sampling/importance_sampling_ratio/min": 0.999430775642395, "sampling/importance_sampling_ratio/mean": 1.0001609325408936, "sampling/importance_sampling_ratio/max": 1.0185940265655518, "entropy": 0.001144394394941628, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.49929603934288025, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.49929603934288025, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1016.0} {"timestamp_utc": "2026-04-11T21:34:46Z", "mode": "train", "global_step": 1017, "epoch": 0.03927247451343837, "loss": 0.0998, "grad_norm": 5.0361127853393555, "learning_rate": 6.921212121212122e-06, "num_tokens": 2192001.0, "completions/mean_length": 94.625, "completions/min_length": 90.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.625, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9968419075012207, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.4375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.43611833453178406, "rewards/total_composite/std": 0.1762184202671051, "reward": 0.43611833453178406, "reward_std": 0.1762184202671051, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003153349505737424, "sampling/sampling_logp_difference/max": 0.7861666679382324, "sampling/importance_sampling_ratio/min": 0.45558789372444153, "sampling/importance_sampling_ratio/mean": 1.000700831413269, "sampling/importance_sampling_ratio/max": 1.220620036125183, "entropy": 0.011739744455553591, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.43611833453178406, "reward_meter_mean": 0.9968419075012207, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.4375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.43611833453178406, "reward_total_composite_std": 0.1762184202671051, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1017.0} {"timestamp_utc": "2026-04-11T21:34:56Z", "mode": "train", "global_step": 1018, "epoch": 0.039311090515909794, "loss": -0.0074, "grad_norm": 0.8842256665229797, "learning_rate": 6.918181818181818e-06, "num_tokens": 2194475.0, "completions/mean_length": 324.25, "completions/min_length": 191.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 211.60000610351562, "completions/min_terminated_length": 191.0, "completions/max_terminated_length": 234.0, "rewards/meter/mean": 0.25342705845832825, "rewards/meter/std": 0.22087812423706055, "rewards/count_adherence/mean": 0.32500001788139343, "rewards/count_adherence/std": 0.10350984334945679, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.04614783823490143, "rewards/total_composite/std": 0.06616245955228806, "reward": 0.04614783823490143, "reward_std": 0.06616245955228806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013864245265722275, "sampling/sampling_logp_difference/max": 1.2185118198394775, "sampling/importance_sampling_ratio/min": 0.2956698536872864, "sampling/importance_sampling_ratio/mean": 1.0000169277191162, "sampling/importance_sampling_ratio/max": 1.663276195526123, "entropy": 0.0381024784874171, "clip_ratio/low_mean": 0.005775065976195037, "clip_ratio/low_min": 0.005775065976195037, "clip_ratio/high_mean": 0.002314814832061529, "clip_ratio/high_max": 0.002314814832061529, "clip_ratio/region_mean": 0.008089880808256567, "reward_total_mean": 0.04614783823490143, "reward_meter_mean": 0.25342705845832825, "reward_meter_std": 0.22087812423706055, "reward_count_adherence_mean": 0.32500001788139343, "reward_count_adherence_std": 0.10350984334945679, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.04614783823490143, "reward_total_composite_std": 0.06616245955228806, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1018.0} {"timestamp_utc": "2026-04-11T21:35:05Z", "mode": "train", "global_step": 1019, "epoch": 0.03934970651838122, "loss": 0.0137, "grad_norm": 0.47353506088256836, "learning_rate": 6.915151515151515e-06, "num_tokens": 2199173.0, "completions/mean_length": 408.25, "completions/min_length": 400.0, "completions/max_length": 426.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 408.25, "completions/min_terminated_length": 400.0, "completions/max_terminated_length": 426.0, "rewards/meter/mean": 0.9946007132530212, "rewards/meter/std": 0.0016429515089839697, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6630671620368958, "rewards/total_composite/std": 0.0010952905286103487, "reward": 0.6630671620368958, "reward_std": 0.0010952960001304746, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003522154875099659, "sampling/sampling_logp_difference/max": 1.055159568786621, "sampling/importance_sampling_ratio/min": 0.348136842250824, "sampling/importance_sampling_ratio/mean": 1.0010110139846802, "sampling/importance_sampling_ratio/max": 1.473246455192566, "entropy": 0.01753362553426996, "clip_ratio/low_mean": 0.0015083312755450606, "clip_ratio/low_min": 0.0015083312755450606, "clip_ratio/high_mean": 0.0009351620683446527, "clip_ratio/high_max": 0.0009351620683446527, "clip_ratio/region_mean": 0.0024434933438897133, "reward_total_mean": 0.6630671620368958, "reward_meter_mean": 0.9946007132530212, "reward_meter_std": 0.0016429515089839697, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6630671620368958, "reward_total_composite_std": 0.0010952905286103487, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1019.0} {"timestamp_utc": "2026-04-11T21:35:11Z", "mode": "train", "global_step": 1020, "epoch": 0.03938832252085264, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.912121212121212e-06, "num_tokens": 2201269.0, "completions/mean_length": 97.0, "completions/min_length": 97.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.0, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9951313734054565, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.49756568670272827, "rewards/total_composite/std": 0.0, "reward": 0.49756568670272827, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0007957889465615153, "sampling/sampling_logp_difference/max": 0.07173186540603638, "sampling/importance_sampling_ratio/min": 0.9307804107666016, "sampling/importance_sampling_ratio/mean": 1.0003042221069336, "sampling/importance_sampling_ratio/max": 1.0364848375320435, "entropy": 0.007352605927735567, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.49756568670272827, "reward_meter_mean": 0.9951313734054565, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.49756568670272827, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1020.0} {"timestamp_utc": "2026-04-11T21:35:18Z", "mode": "train", "global_step": 1021, "epoch": 0.039426938523324066, "loss": 0.1144, "grad_norm": 5.586789608001709, "learning_rate": 6.90909090909091e-06, "num_tokens": 2203382.0, "completions/mean_length": 107.125, "completions/min_length": 97.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.125, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9948334693908691, "rewards/meter/std": 0.0006093190750107169, "rewards/count_adherence/mean": 0.375, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.37302374839782715, "rewards/total_composite/std": 0.23023545742034912, "reward": 0.37302374839782715, "reward_std": 0.23023544251918793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008710247464478016, "sampling/sampling_logp_difference/max": 0.9759321212768555, "sampling/importance_sampling_ratio/min": 0.3768409192562103, "sampling/importance_sampling_ratio/mean": 0.9977646470069885, "sampling/importance_sampling_ratio/max": 1.2892647981643677, "entropy": 0.031326725613325834, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007536343880929053, "clip_ratio/high_max": 0.007536343880929053, "clip_ratio/region_mean": 0.007536343880929053, "reward_total_mean": 0.37302374839782715, "reward_meter_mean": 0.9948334693908691, "reward_meter_std": 0.0006093190750107169, "reward_count_adherence_mean": 0.375, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.37302374839782715, "reward_total_composite_std": 0.23023545742034912, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1021.0} {"timestamp_utc": "2026-04-11T21:35:23Z", "mode": "train", "global_step": 1022, "epoch": 0.03946555452579549, "loss": -0.024, "grad_norm": 5.417184352874756, "learning_rate": 6.906060606060606e-06, "num_tokens": 2205244.0, "completions/mean_length": 78.75, "completions/min_length": 75.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.75, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.3565239906311035, "rewards/meter/std": 0.355892151594162, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.17826199531555176, "rewards/total_composite/std": 0.177946075797081, "reward": 0.17826199531555176, "reward_std": 0.1779460608959198, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030668344348669052, "sampling/sampling_logp_difference/max": 4.095977783203125, "sampling/importance_sampling_ratio/min": 0.016639469191432, "sampling/importance_sampling_ratio/mean": 1.0034855604171753, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10641406895592809, "clip_ratio/low_mean": 0.016281440039165318, "clip_ratio/low_min": 0.016281440039165318, "clip_ratio/high_mean": 0.010758944670669734, "clip_ratio/high_max": 0.010758944670669734, "clip_ratio/region_mean": 0.027040384709835052, "reward_total_mean": 0.17826199531555176, "reward_meter_mean": 0.3565239906311035, "reward_meter_std": 0.355892151594162, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.17826199531555176, "reward_total_composite_std": 0.177946075797081, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1022.0} {"timestamp_utc": "2026-04-11T21:35:28Z", "mode": "train", "global_step": 1023, "epoch": 0.039504170528266914, "loss": 0.0704, "grad_norm": 2.357210636138916, "learning_rate": 6.903030303030304e-06, "num_tokens": 2207420.0, "completions/mean_length": 93.0, "completions/min_length": 81.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.0, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9951544404029846, "rewards/meter/std": 6.528547237394378e-05, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6219686269760132, "rewards/total_composite/std": 0.23032104969024658, "reward": 0.6219686269760132, "reward_std": 0.2303210347890854, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004541828762739897, "sampling/sampling_logp_difference/max": 1.1389760971069336, "sampling/importance_sampling_ratio/min": 0.5130444765090942, "sampling/importance_sampling_ratio/mean": 1.0001682043075562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.007780839689075947, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0015432098880410194, "clip_ratio/high_max": 0.0015432098880410194, "clip_ratio/region_mean": 0.0015432098880410194, "reward_total_mean": 0.6219686269760132, "reward_meter_mean": 0.9951544404029846, "reward_meter_std": 6.528547237394378e-05, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6219686269760132, "reward_total_composite_std": 0.23032104969024658, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1023.0} {"timestamp_utc": "2026-04-11T21:35:38Z", "mode": "train", "global_step": 1024, "epoch": 0.03954278653073834, "loss": 0.0561, "grad_norm": 5.333614826202393, "learning_rate": 6.9e-06, "num_tokens": 2210634.0, "completions/mean_length": 288.75, "completions/min_length": 253.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 256.8571472167969, "completions/min_terminated_length": 253.0, "completions/max_terminated_length": 260.0, "rewards/meter/mean": 0.880677342414856, "rewards/meter/std": 0.23732006549835205, "rewards/count_adherence/mean": 0.42500001192092896, "rewards/count_adherence/std": 0.0707106813788414, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3606422543525696, "rewards/total_composite/std": 0.07344617694616318, "reward": 0.3606422543525696, "reward_std": 0.07344616949558258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0037221608217805624, "sampling/sampling_logp_difference/max": 1.0087032318115234, "sampling/importance_sampling_ratio/min": 0.3646915853023529, "sampling/importance_sampling_ratio/mean": 0.9999430775642395, "sampling/importance_sampling_ratio/max": 1.8063973188400269, "entropy": 0.009922248020302504, "clip_ratio/low_mean": 0.001936378888785839, "clip_ratio/low_min": 0.001936378888785839, "clip_ratio/high_mean": 0.00048638132284395397, "clip_ratio/high_max": 0.00048638132284395397, "clip_ratio/region_mean": 0.002422760211629793, "reward_total_mean": 0.3606422543525696, "reward_meter_mean": 0.880677342414856, "reward_meter_std": 0.23732006549835205, "reward_count_adherence_mean": 0.42500001192092896, "reward_count_adherence_std": 0.0707106813788414, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3606422543525696, "reward_total_composite_std": 0.07344617694616318, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1024.0} {"timestamp_utc": "2026-04-11T21:35:48Z", "mode": "train", "global_step": 1025, "epoch": 0.03958140253320976, "loss": -0.0962, "grad_norm": 2.2874205112457275, "learning_rate": 6.896969696969697e-06, "num_tokens": 2214561.0, "completions/mean_length": 330.875, "completions/min_length": 281.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 305.0, "completions/min_terminated_length": 281.0, "completions/max_terminated_length": 343.0, "rewards/meter/mean": 0.20773091912269592, "rewards/meter/std": 0.1968511939048767, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.10858245193958282, "rewards/total_composite/std": 0.1236080750823021, "reward": 0.10858245193958282, "reward_std": 0.1236080750823021, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012201424688100815, "sampling/sampling_logp_difference/max": 9.05886173248291, "sampling/importance_sampling_ratio/min": 0.0001163553461083211, "sampling/importance_sampling_ratio/mean": 1.0015368461608887, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02456760942004621, "clip_ratio/low_mean": 0.00296406022971496, "clip_ratio/low_min": 0.00296406022971496, "clip_ratio/high_mean": 0.0015677891788072884, "clip_ratio/high_max": 0.0015677891788072884, "clip_ratio/region_mean": 0.004531849408522248, "reward_total_mean": 0.10858245193958282, "reward_meter_mean": 0.20773091912269592, "reward_meter_std": 0.1968511939048767, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.10858245193958282, "reward_total_composite_std": 0.1236080750823021, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1025.0} {"timestamp_utc": "2026-04-11T21:35:53Z", "mode": "train", "global_step": 1026, "epoch": 0.03962001853568119, "loss": 0.011, "grad_norm": 6.932950973510742, "learning_rate": 6.893939393939395e-06, "num_tokens": 2216302.0, "completions/mean_length": 63.625, "completions/min_length": 63.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9958137273788452, "rewards/meter/std": 0.0004962231614626944, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9958137273788452, "rewards/total_composite/std": 0.0004962231614626944, "reward": 0.9958137273788452, "reward_std": 0.0004962170496582985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007021648343652487, "sampling/sampling_logp_difference/max": 0.7024054527282715, "sampling/importance_sampling_ratio/min": 0.49539220333099365, "sampling/importance_sampling_ratio/mean": 0.9998581409454346, "sampling/importance_sampling_ratio/max": 1.4773842096328735, "entropy": 0.016864305711351335, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/region_mean": 0.003937252098694444, "reward_total_mean": 0.9958137273788452, "reward_meter_mean": 0.9958137273788452, "reward_meter_std": 0.0004962231614626944, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9958137273788452, "reward_total_composite_std": 0.0004962231614626944, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1026.0} {"timestamp_utc": "2026-04-11T21:35:58Z", "mode": "train", "global_step": 1027, "epoch": 0.03965863453815261, "loss": 0.003, "grad_norm": 2.410141944885254, "learning_rate": 6.890909090909092e-06, "num_tokens": 2218001.0, "completions/mean_length": 65.375, "completions/min_length": 61.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.375, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9960290193557739, "rewards/meter/std": 0.002299173967912793, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9960290193557739, "rewards/total_composite/std": 0.002299173967912793, "reward": 0.9960290193557739, "reward_std": 0.002299182815477252, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01110776886343956, "sampling/sampling_logp_difference/max": 1.0982933044433594, "sampling/importance_sampling_ratio/min": 0.35885050892829895, "sampling/importance_sampling_ratio/mean": 1.0042622089385986, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.014273899607360363, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0018939394503831863, "reward_total_mean": 0.9960290193557739, "reward_meter_mean": 0.9960290193557739, "reward_meter_std": 0.002299173967912793, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9960290193557739, "reward_total_composite_std": 0.002299173967912793, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1027.0} {"timestamp_utc": "2026-04-11T21:36:04Z", "mode": "train", "global_step": 1028, "epoch": 0.039697250540624035, "loss": -0.0237, "grad_norm": 3.056302309036255, "learning_rate": 6.887878787878789e-06, "num_tokens": 2220288.0, "completions/mean_length": 128.875, "completions/min_length": 121.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.875, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.9919188022613525, "rewards/meter/std": 0.007901563309133053, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6612792015075684, "rewards/total_composite/std": 0.005267724394798279, "reward": 0.6612792015075684, "reward_std": 0.005267728585749865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007906788028776646, "sampling/sampling_logp_difference/max": 2.279299259185791, "sampling/importance_sampling_ratio/min": 0.10235590487718582, "sampling/importance_sampling_ratio/mean": 0.9983795881271362, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.01950788137037307, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.003826094383839518, "clip_ratio/high_max": 0.003826094383839518, "clip_ratio/region_mean": 0.003826094383839518, "reward_total_mean": 0.6612792015075684, "reward_meter_mean": 0.9919188022613525, "reward_meter_std": 0.007901563309133053, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6612792015075684, "reward_total_composite_std": 0.005267724394798279, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1028.0} {"timestamp_utc": "2026-04-11T21:36:08Z", "mode": "train", "global_step": 1029, "epoch": 0.03973586654309546, "loss": 0.0026, "grad_norm": 2.344759225845337, "learning_rate": 6.8848484848484854e-06, "num_tokens": 2222027.0, "completions/mean_length": 50.375, "completions/min_length": 50.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9583595991134644, "rewards/meter/std": 0.0016159254591912031, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9583595991134644, "rewards/total_composite/std": 0.0016159254591912031, "reward": 0.9583595991134644, "reward_std": 0.0016159284859895706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008701321668922901, "sampling/sampling_logp_difference/max": 1.1940813064575195, "sampling/importance_sampling_ratio/min": 0.3029821813106537, "sampling/importance_sampling_ratio/mean": 0.9988434910774231, "sampling/importance_sampling_ratio/max": 1.1769939661026, "entropy": 0.03183695743791759, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/high_mean": 0.0024999999441206455, "clip_ratio/high_max": 0.0024999999441206455, "clip_ratio/region_mean": 0.007307692430913448, "reward_total_mean": 0.9583595991134644, "reward_meter_mean": 0.9583595991134644, "reward_meter_std": 0.0016159254591912031, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9583595991134644, "reward_total_composite_std": 0.0016159254591912031, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1029.0} {"timestamp_utc": "2026-04-11T21:36:13Z", "mode": "train", "global_step": 1030, "epoch": 0.03977448254556688, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.881818181818183e-06, "num_tokens": 2223933.0, "completions/mean_length": 74.25, "completions/min_length": 74.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.25, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9686469435691833, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9686469435691833, "rewards/total_composite/std": 0.0, "reward": 0.9686469435691833, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.004110315814614296, "sampling/sampling_logp_difference/max": 0.5173161029815674, "sampling/importance_sampling_ratio/min": 0.5961183309555054, "sampling/importance_sampling_ratio/mean": 0.9992336630821228, "sampling/importance_sampling_ratio/max": 1.095316767692566, "entropy": 0.020718287560157478, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9686469435691833, "reward_meter_mean": 0.9686469435691833, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9686469435691833, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1030.0} {"timestamp_utc": "2026-04-11T21:36:21Z", "mode": "train", "global_step": 1031, "epoch": 0.03981309854803831, "loss": -0.0819, "grad_norm": 2.833019733428955, "learning_rate": 6.878787878787879e-06, "num_tokens": 2227258.0, "completions/mean_length": 228.625, "completions/min_length": 201.0, "completions/max_length": 258.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 228.625, "completions/min_terminated_length": 201.0, "completions/max_terminated_length": 258.0, "rewards/meter/mean": 0.6067942976951599, "rewards/meter/std": 0.49607783555984497, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.07715165615081787, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4459972083568573, "rewards/total_composite/std": 0.37391820549964905, "reward": 0.4459972083568573, "reward_std": 0.37391817569732666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007401937153190374, "sampling/sampling_logp_difference/max": 1.5288528203964233, "sampling/importance_sampling_ratio/min": 0.21678423881530762, "sampling/importance_sampling_ratio/mean": 0.9979560375213623, "sampling/importance_sampling_ratio/max": 1.5826334953308105, "entropy": 0.012700279825367033, "clip_ratio/low_mean": 0.001243781065568328, "clip_ratio/low_min": 0.001243781065568328, "clip_ratio/high_mean": 0.004176991351414472, "clip_ratio/high_max": 0.004176991351414472, "clip_ratio/region_mean": 0.0054207724169828, "reward_total_mean": 0.4459972083568573, "reward_meter_mean": 0.6067942976951599, "reward_meter_std": 0.49607783555984497, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.07715165615081787, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4459972083568573, "reward_total_composite_std": 0.37391820549964905, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1031.0} {"timestamp_utc": "2026-04-11T21:36:25Z", "mode": "train", "global_step": 1032, "epoch": 0.03985171455050973, "loss": -0.1231, "grad_norm": 12.452238082885742, "learning_rate": 6.875757575757576e-06, "num_tokens": 2228997.0, "completions/mean_length": 44.375, "completions/min_length": 38.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.375, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.944683313369751, "rewards/meter/std": 0.015247092582285404, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.26726123690605164, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7120780944824219, "rewards/total_composite/std": 0.263912558555603, "reward": 0.7120780944824219, "reward_std": 0.263912558555603, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012518586590886116, "sampling/sampling_logp_difference/max": 1.3683909177780151, "sampling/importance_sampling_ratio/min": 0.2545161843299866, "sampling/importance_sampling_ratio/mean": 1.00504469871521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03652537357993424, "clip_ratio/low_mean": 0.00657894741743803, "clip_ratio/low_min": 0.00657894741743803, "clip_ratio/high_mean": 0.007401960901916027, "clip_ratio/high_max": 0.007401960901916027, "clip_ratio/region_mean": 0.013980908319354057, "reward_total_mean": 0.7120780944824219, "reward_meter_mean": 0.944683313369751, "reward_meter_std": 0.015247092582285404, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.26726123690605164, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7120780944824219, "reward_total_composite_std": 0.263912558555603, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1032.0} {"timestamp_utc": "2026-04-11T21:36:36Z", "mode": "train", "global_step": 1033, "epoch": 0.039890330552981156, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.872727272727273e-06, "num_tokens": 2230693.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9922091364860535, "rewards/meter/std": 0.0014619192807003856, "rewards/count_adherence/mean": 0.2083333432674408, "rewards/count_adherence/std": 0.07715167850255966, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.20674961805343628, "rewards/total_composite/std": 0.07672107964754105, "reward": 0.20674961805343628, "reward_std": 0.07672107964754105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.20674961805343628, "reward_meter_mean": 0.9922091364860535, "reward_meter_std": 0.0014619192807003856, "reward_count_adherence_mean": 0.2083333432674408, "reward_count_adherence_std": 0.07715167850255966, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.20674961805343628, "reward_total_composite_std": 0.07672107964754105, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1033.0} {"timestamp_utc": "2026-04-11T21:36:42Z", "mode": "train", "global_step": 1034, "epoch": 0.03992894655545258, "loss": -0.001, "grad_norm": 3.635939598083496, "learning_rate": 6.869696969696971e-06, "num_tokens": 2233601.0, "completions/mean_length": 159.5, "completions/min_length": 141.0, "completions/max_length": 177.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.5, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 177.0, "rewards/meter/mean": 0.4298320412635803, "rewards/meter/std": 0.2531501054763794, "rewards/count_adherence/mean": 0.59375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.2574954032897949, "rewards/total_composite/std": 0.16803935170173645, "reward": 0.2574954032897949, "reward_std": 0.16803935170173645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017568040639162064, "sampling/sampling_logp_difference/max": 1.0464534759521484, "sampling/importance_sampling_ratio/min": 0.3511810302734375, "sampling/importance_sampling_ratio/mean": 1.0035357475280762, "sampling/importance_sampling_ratio/max": 1.6600509881973267, "entropy": 0.1108170528896153, "clip_ratio/low_mean": 0.009340384451206774, "clip_ratio/low_min": 0.009340384451206774, "clip_ratio/high_mean": 0.0039124294416978955, "clip_ratio/high_max": 0.0039124294416978955, "clip_ratio/region_mean": 0.013252813892904669, "reward_total_mean": 0.2574954032897949, "reward_meter_mean": 0.4298320412635803, "reward_meter_std": 0.2531501054763794, "reward_count_adherence_mean": 0.59375, "reward_count_adherence_std": 0.12938730418682098, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.2574954032897949, "reward_total_composite_std": 0.16803935170173645, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1034.0} {"timestamp_utc": "2026-04-11T21:36:47Z", "mode": "train", "global_step": 1035, "epoch": 0.039967562557924004, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.866666666666667e-06, "num_tokens": 2235009.0, "completions/mean_length": 31.0, "completions/min_length": 31.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00018558392184786499, "sampling/sampling_logp_difference/max": 0.006259385496377945, "sampling/importance_sampling_ratio/min": 0.9999586343765259, "sampling/importance_sampling_ratio/mean": 1.000185251235962, "sampling/importance_sampling_ratio/max": 1.0062791109085083, "entropy": 0.002012173834373243, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1035.0} {"timestamp_utc": "2026-04-11T21:36:52Z", "mode": "train", "global_step": 1036, "epoch": 0.04000617856039543, "loss": -0.0019, "grad_norm": 3.399261951446533, "learning_rate": 6.8636363636363645e-06, "num_tokens": 2237085.0, "completions/mean_length": 90.5, "completions/min_length": 90.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.5, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9966752529144287, "rewards/meter/std": 0.00017812939768191427, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9966752529144287, "rewards/total_composite/std": 0.00017812939768191427, "reward": 0.9966752529144287, "reward_std": 0.00017812938312999904, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0035630161873996258, "sampling/sampling_logp_difference/max": 1.0746439695358276, "sampling/importance_sampling_ratio/min": 0.3414193093776703, "sampling/importance_sampling_ratio/mean": 0.999756395816803, "sampling/importance_sampling_ratio/max": 1.3620209693908691, "entropy": 0.00802014273358509, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0013736264081671834, "clip_ratio/high_max": 0.0013736264081671834, "clip_ratio/region_mean": 0.0013736264081671834, "reward_total_mean": 0.9966752529144287, "reward_meter_mean": 0.9966752529144287, "reward_meter_std": 0.00017812939768191427, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9966752529144287, "reward_total_composite_std": 0.00017812939768191427, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1036.0} {"timestamp_utc": "2026-04-11T21:36:56Z", "mode": "train", "global_step": 1037, "epoch": 0.04004479456286685, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.860606060606061e-06, "num_tokens": 2238973.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002364885003771633, "sampling/sampling_logp_difference/max": 0.016218479722738266, "sampling/importance_sampling_ratio/min": 0.9962993264198303, "sampling/importance_sampling_ratio/mean": 1.0002158880233765, "sampling/importance_sampling_ratio/max": 1.0163507461547852, "entropy": 0.0025363014137838036, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1037.0} {"timestamp_utc": "2026-04-11T21:37:01Z", "mode": "train", "global_step": 1038, "epoch": 0.040083410565338276, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.857575757575758e-06, "num_tokens": 2241173.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00012717192294076085, "sampling/sampling_logp_difference/max": 0.014000311493873596, "sampling/importance_sampling_ratio/min": 0.9999371767044067, "sampling/importance_sampling_ratio/mean": 1.0001276731491089, "sampling/importance_sampling_ratio/max": 1.0140987634658813, "entropy": 0.001326097029959783, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1038.0} {"timestamp_utc": "2026-04-11T21:37:08Z", "mode": "train", "global_step": 1039, "epoch": 0.0401220265678097, "loss": 0.0352, "grad_norm": 1.2112030982971191, "learning_rate": 6.854545454545455e-06, "num_tokens": 2243798.0, "completions/mean_length": 161.125, "completions/min_length": 145.0, "completions/max_length": 187.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 161.125, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 187.0, "rewards/meter/mean": 0.9395418167114258, "rewards/meter/std": 0.06988690048456192, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6746968626976013, "rewards/total_composite/std": 0.0945713222026825, "reward": 0.6746968626976013, "reward_std": 0.0945713222026825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006402490194886923, "sampling/sampling_logp_difference/max": 1.0784833431243896, "sampling/importance_sampling_ratio/min": 0.3401109576225281, "sampling/importance_sampling_ratio/mean": 0.9998781085014343, "sampling/importance_sampling_ratio/max": 1.328989863395691, "entropy": 0.02901237178593874, "clip_ratio/low_mean": 0.0036028079339303076, "clip_ratio/low_min": 0.0036028079339303076, "clip_ratio/high_mean": 0.002329192589968443, "clip_ratio/high_max": 0.002329192589968443, "clip_ratio/region_mean": 0.0059320005238987505, "reward_total_mean": 0.6746968626976013, "reward_meter_mean": 0.9395418167114258, "reward_meter_std": 0.06988690048456192, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6746968626976013, "reward_total_composite_std": 0.0945713222026825, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1039.0} {"timestamp_utc": "2026-04-11T21:37:18Z", "mode": "train", "global_step": 1040, "epoch": 0.040160642570281124, "loss": -0.1429, "grad_norm": 0.6260462999343872, "learning_rate": 6.851515151515153e-06, "num_tokens": 2245773.0, "completions/mean_length": 402.875, "completions/min_length": 81.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 221.0, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 485.0, "rewards/meter/mean": 0.9941596984863281, "rewards/meter/std": 0.0014686459908261895, "rewards/count_adherence/mean": 0.5416666865348816, "rewards/count_adherence/std": 0.24800795316696167, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5385435819625854, "rewards/total_composite/std": 0.24677012860774994, "reward": 0.5385435819625854, "reward_std": 0.24677012860774994, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006288351025432348, "sampling/sampling_logp_difference/max": 1.5349009037017822, "sampling/importance_sampling_ratio/min": 0.21547703444957733, "sampling/importance_sampling_ratio/mean": 0.9992893934249878, "sampling/importance_sampling_ratio/max": 1.5936279296875, "entropy": 0.007368071412201971, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.002577319508418441, "clip_ratio/high_max": 0.002577319508418441, "clip_ratio/region_mean": 0.002577319508418441, "reward_total_mean": 0.5385435819625854, "reward_meter_mean": 0.9941596984863281, "reward_meter_std": 0.0014686459908261895, "reward_count_adherence_mean": 0.5416666865348816, "reward_count_adherence_std": 0.24800795316696167, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5385435819625854, "reward_total_composite_std": 0.24677012860774994, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1040.0} {"timestamp_utc": "2026-04-11T21:37:22Z", "mode": "train", "global_step": 1041, "epoch": 0.04019925857275255, "loss": -0.0013, "grad_norm": 1.3498170375823975, "learning_rate": 6.848484848484849e-06, "num_tokens": 2247341.0, "completions/mean_length": 33.0, "completions/min_length": 33.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9939789772033691, "rewards/meter/std": 0.0011234743287786841, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9939789772033691, "rewards/total_composite/std": 0.0011234743287786841, "reward": 0.9939789772033691, "reward_std": 0.001123474445194006, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005430762656033039, "sampling/sampling_logp_difference/max": 0.5017609596252441, "sampling/importance_sampling_ratio/min": 0.7880626320838928, "sampling/importance_sampling_ratio/mean": 1.0029045343399048, "sampling/importance_sampling_ratio/max": 1.6516271829605103, "entropy": 0.02535962895490229, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.007575757801532745, "reward_total_mean": 0.9939789772033691, "reward_meter_mean": 0.9939789772033691, "reward_meter_std": 0.0011234743287786841, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9939789772033691, "reward_total_composite_std": 0.0011234743287786841, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1041.0} {"timestamp_utc": "2026-04-11T21:37:28Z", "mode": "train", "global_step": 1042, "epoch": 0.04023787457522397, "loss": -0.0174, "grad_norm": 1.30880868434906, "learning_rate": 6.845454545454546e-06, "num_tokens": 2249989.0, "completions/mean_length": 146.0, "completions/min_length": 145.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.0, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.9235571622848511, "rewards/meter/std": 0.010089955292642117, "rewards/count_adherence/mean": 0.4000000059604645, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3694228529930115, "rewards/total_composite/std": 0.0040359823033213615, "reward": 0.3694228529930115, "reward_std": 0.0040359823033213615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002169681014493108, "sampling/sampling_logp_difference/max": 0.4037449359893799, "sampling/importance_sampling_ratio/min": 0.7361330986022949, "sampling/importance_sampling_ratio/mean": 1.0009979009628296, "sampling/importance_sampling_ratio/max": 1.4974219799041748, "entropy": 0.010642981505952775, "clip_ratio/low_mean": 0.003448275849223137, "clip_ratio/low_min": 0.003448275849223137, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.003448275849223137, "reward_total_mean": 0.3694228529930115, "reward_meter_mean": 0.9235571622848511, "reward_meter_std": 0.010089955292642117, "reward_count_adherence_mean": 0.4000000059604645, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3694228529930115, "reward_total_composite_std": 0.0040359823033213615, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1042.0} {"timestamp_utc": "2026-04-11T21:37:33Z", "mode": "train", "global_step": 1043, "epoch": 0.0402764905776954, "loss": -0.1228, "grad_norm": 3.558645486831665, "learning_rate": 6.842424242424243e-06, "num_tokens": 2251649.0, "completions/mean_length": 57.5, "completions/min_length": 49.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9965112805366516, "rewards/meter/std": 4.913831435260363e-05, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.26726123690605164, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7473846673965454, "rewards/total_composite/std": 0.266332745552063, "reward": 0.7473846673965454, "reward_std": 0.2663327157497406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012323648668825626, "sampling/sampling_logp_difference/max": 0.5443525314331055, "sampling/importance_sampling_ratio/min": 0.5802173614501953, "sampling/importance_sampling_ratio/mean": 1.0003561973571777, "sampling/importance_sampling_ratio/max": 1.3213616609573364, "entropy": 0.08020356902852654, "clip_ratio/low_mean": 0.009819021914154291, "clip_ratio/low_min": 0.009819021914154291, "clip_ratio/high_mean": 0.007692307699471712, "clip_ratio/high_max": 0.007692307699471712, "clip_ratio/region_mean": 0.017511329613626003, "reward_total_mean": 0.7473846673965454, "reward_meter_mean": 0.9965112805366516, "reward_meter_std": 4.913831435260363e-05, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.26726123690605164, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7473846673965454, "reward_total_composite_std": 0.266332745552063, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1043.0} {"timestamp_utc": "2026-04-11T21:37:37Z", "mode": "train", "global_step": 1044, "epoch": 0.04031510658016682, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.83939393939394e-06, "num_tokens": 2253361.0, "completions/mean_length": 46.0, "completions/min_length": 46.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9968419075012207, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.49842095375061035, "rewards/total_composite/std": 0.0, "reward": 0.49842095375061035, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00031323276925832033, "sampling/sampling_logp_difference/max": 0.003529260866343975, "sampling/importance_sampling_ratio/min": 0.9989622235298157, "sampling/importance_sampling_ratio/mean": 1.000293254852295, "sampling/importance_sampling_ratio/max": 1.003535509109497, "entropy": 0.0027704476087819785, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.49842095375061035, "reward_meter_mean": 0.9968419075012207, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.49842095375061035, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1044.0} {"timestamp_utc": "2026-04-11T21:37:42Z", "mode": "train", "global_step": 1045, "epoch": 0.040353722582638245, "loss": -0.0306, "grad_norm": 2.4948503971099854, "learning_rate": 6.8363636363636364e-06, "num_tokens": 2255287.0, "completions/mean_length": 63.75, "completions/min_length": 59.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.75, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9814295768737793, "rewards/meter/std": 0.03899209573864937, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9814295768737793, "rewards/total_composite/std": 0.03899209573864937, "reward": 0.9814295768737793, "reward_std": 0.038992080837488174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00862820539623499, "sampling/sampling_logp_difference/max": 0.8584649562835693, "sampling/importance_sampling_ratio/min": 0.4238121807575226, "sampling/importance_sampling_ratio/mean": 0.998173713684082, "sampling/importance_sampling_ratio/max": 1.098639965057373, "entropy": 0.030405770405195653, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.003846153849735856, "clip_ratio/high_max": 0.003846153849735856, "clip_ratio/region_mean": 0.003846153849735856, "reward_total_mean": 0.9814295768737793, "reward_meter_mean": 0.9814295768737793, "reward_meter_std": 0.03899209573864937, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9814295768737793, "reward_total_composite_std": 0.03899209573864937, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1045.0} {"timestamp_utc": "2026-04-11T21:37:46Z", "mode": "train", "global_step": 1046, "epoch": 0.04039233858510967, "loss": 0.0163, "grad_norm": 2.792025089263916, "learning_rate": 6.833333333333334e-06, "num_tokens": 2256868.0, "completions/mean_length": 45.625, "completions/min_length": 43.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.625, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.997046709060669, "rewards/meter/std": 0.0005792452138848603, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4985233545303345, "rewards/total_composite/std": 0.00028962260694243014, "reward": 0.4985233545303345, "reward_std": 0.00028962711803615093, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003296646988019347, "sampling/sampling_logp_difference/max": 0.7852382659912109, "sampling/importance_sampling_ratio/min": 0.4560110569000244, "sampling/importance_sampling_ratio/mean": 0.9996577501296997, "sampling/importance_sampling_ratio/max": 1.2223551273345947, "entropy": 0.013065017032204196, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0029069767333567142, "clip_ratio/high_max": 0.0029069767333567142, "clip_ratio/region_mean": 0.0029069767333567142, "reward_total_mean": 0.4985233545303345, "reward_meter_mean": 0.997046709060669, "reward_meter_std": 0.0005792452138848603, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4985233545303345, "reward_total_composite_std": 0.00028962260694243014, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1046.0} {"timestamp_utc": "2026-04-11T21:37:51Z", "mode": "train", "global_step": 1047, "epoch": 0.04043095458758109, "loss": 0.0241, "grad_norm": 7.9632158279418945, "learning_rate": 6.83030303030303e-06, "num_tokens": 2258120.0, "completions/mean_length": 33.5, "completions/min_length": 33.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9939615726470947, "rewards/meter/std": 0.000965635001193732, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9939615726470947, "rewards/total_composite/std": 0.000965635001193732, "reward": 0.9939615726470947, "reward_std": 0.0009656234178692102, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00958024151623249, "sampling/sampling_logp_difference/max": 0.5317777395248413, "sampling/importance_sampling_ratio/min": 0.6065818667411804, "sampling/importance_sampling_ratio/mean": 1.002090573310852, "sampling/importance_sampling_ratio/max": 1.7019551992416382, "entropy": 0.032024843618273735, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/region_mean": 0.01093073608353734, "reward_total_mean": 0.9939615726470947, "reward_meter_mean": 0.9939615726470947, "reward_meter_std": 0.000965635001193732, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9939615726470947, "reward_total_composite_std": 0.000965635001193732, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1047.0} {"timestamp_utc": "2026-04-11T21:37:56Z", "mode": "train", "global_step": 1048, "epoch": 0.04046957059005252, "loss": -0.0071, "grad_norm": 3.083638906478882, "learning_rate": 6.827272727272728e-06, "num_tokens": 2259982.0, "completions/mean_length": 65.75, "completions/min_length": 61.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9918305277824402, "rewards/meter/std": 0.012018299661576748, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9918305277824402, "rewards/total_composite/std": 0.012018299661576748, "reward": 0.9918305277824402, "reward_std": 0.012018297798931599, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011509421281516552, "sampling/sampling_logp_difference/max": 0.996147632598877, "sampling/importance_sampling_ratio/min": 0.36929938197135925, "sampling/importance_sampling_ratio/mean": 1.0007789134979248, "sampling/importance_sampling_ratio/max": 1.513947606086731, "entropy": 0.06583833554759622, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.0071314104134216905, "clip_ratio/high_max": 0.0071314104134216905, "clip_ratio/region_mean": 0.01103766041342169, "reward_total_mean": 0.9918305277824402, "reward_meter_mean": 0.9918305277824402, "reward_meter_std": 0.012018299661576748, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9918305277824402, "reward_total_composite_std": 0.012018299661576748, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1048.0} {"timestamp_utc": "2026-04-11T21:38:04Z", "mode": "train", "global_step": 1049, "epoch": 0.04050818659252394, "loss": 0.0192, "grad_norm": 2.3660078048706055, "learning_rate": 6.824242424242425e-06, "num_tokens": 2263825.0, "completions/mean_length": 283.375, "completions/min_length": 279.0, "completions/max_length": 291.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 283.375, "completions/min_terminated_length": 279.0, "completions/max_terminated_length": 291.0, "rewards/meter/mean": 0.9826633334159851, "rewards/meter/std": 0.0013277583057060838, "rewards/count_adherence/mean": 0.578125, "rewards/count_adherence/std": 0.06469365209341049, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5680685043334961, "rewards/total_composite/std": 0.06325043737888336, "reward": 0.5680685043334961, "reward_std": 0.06325043737888336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0027596044819802046, "sampling/sampling_logp_difference/max": 1.5669541358947754, "sampling/importance_sampling_ratio/min": 0.20867981016635895, "sampling/importance_sampling_ratio/mean": 1.0000476837158203, "sampling/importance_sampling_ratio/max": 1.6442346572875977, "entropy": 0.006785797595512122, "clip_ratio/low_mean": 0.0012901410227641463, "clip_ratio/low_min": 0.0012901410227641463, "clip_ratio/high_mean": 0.0013440860202535987, "clip_ratio/high_max": 0.0013440860202535987, "clip_ratio/region_mean": 0.002634227043017745, "reward_total_mean": 0.5680685043334961, "reward_meter_mean": 0.9826633334159851, "reward_meter_std": 0.0013277583057060838, "reward_count_adherence_mean": 0.578125, "reward_count_adherence_std": 0.06469365209341049, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5680685043334961, "reward_total_composite_std": 0.06325043737888336, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1049.0} {"timestamp_utc": "2026-04-11T21:38:09Z", "mode": "train", "global_step": 1050, "epoch": 0.040546802594995365, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.821212121212122e-06, "num_tokens": 2265729.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9968419075012207, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9968419075012207, "rewards/total_composite/std": 0.0, "reward": 0.9968419075012207, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0027122944593429565, "sampling/sampling_logp_difference/max": 0.7146163582801819, "sampling/importance_sampling_ratio/min": 0.4893798530101776, "sampling/importance_sampling_ratio/mean": 0.999724805355072, "sampling/importance_sampling_ratio/max": 1.1012705564498901, "entropy": 0.007037240080535412, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9968419075012207, "reward_meter_mean": 0.9968419075012207, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9968419075012207, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1050.0} {"timestamp_utc": "2026-04-11T21:39:35Z", "mode": "eval", "global_step": 1050, "epoch": 0.040546802594995365, "eval_loss": NaN, "eval_runtime": 85.4042, "eval_samples_per_second": 1.218, "eval_steps_per_second": 0.152, "eval_num_tokens": 2265729.0, "eval_completions/mean_length": 241.28846153846155, "eval_completions/min_length": 59.61538461538461, "eval_completions/max_length": 458.0, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/mean_terminated_length": 224.83516869178186, "eval_completions/min_terminated_length": 59.61538461538461, "eval_completions/max_terminated_length": 412.46153846153845, "eval_rewards/meter/mean": 0.6695947761719043, "eval_rewards/meter/std": 0.4209407364519743, "eval_rewards/count_adherence/mean": 0.8243002341343806, "eval_rewards/count_adherence/std": 0.149119944526599, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/total_composite/mean": 0.5442026945260855, "eval_rewards/total_composite/std": 0.3753266856074333, "eval_reward": 0.5442026945260855, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0014539593382953452, "eval_sampling/sampling_logp_difference/max": 0.48829063085409313, "eval_sampling/importance_sampling_ratio/min": 0.6470333154384906, "eval_sampling/importance_sampling_ratio/mean": 1.0002473134260912, "eval_sampling/importance_sampling_ratio/max": 1.2309852104920607, "eval_entropy": 0.009740084463443894, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5442026945260855, "eval_reward_meter_mean": 0.6695947761719043, "eval_reward_meter_std": 0.4209407364519743, "eval_reward_count_adherence_mean": 0.8243002341343806, "eval_reward_count_adherence_std": 0.149119944526599, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_total_composite_mean": 0.5442026945260855, "eval_reward_total_composite_std": 0.3753266856074333, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1050.0} {"timestamp_utc": "2026-04-11T21:39:43Z", "mode": "train", "global_step": 1051, "epoch": 0.04058541859746679, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.818181818181818e-06, "num_tokens": 2267377.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00045387333375401795, "sampling/sampling_logp_difference/max": 0.042022280395030975, "sampling/importance_sampling_ratio/min": 0.9996206760406494, "sampling/importance_sampling_ratio/mean": 1.0004585981369019, "sampling/importance_sampling_ratio/max": 1.0429177284240723, "entropy": 0.004360992228612304, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1051.0} {"timestamp_utc": "2026-04-11T21:39:48Z", "mode": "train", "global_step": 1052, "epoch": 0.040624034599938214, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.8151515151515155e-06, "num_tokens": 2269065.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9968419075012207, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9968419075012207, "rewards/total_composite/std": 0.0, "reward": 0.9968419075012207, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0007426206138916314, "sampling/sampling_logp_difference/max": 0.09673504531383514, "sampling/importance_sampling_ratio/min": 0.9077965617179871, "sampling/importance_sampling_ratio/mean": 1.0001823902130127, "sampling/importance_sampling_ratio/max": 1.0188950300216675, "entropy": 0.004134227987378836, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9968419075012207, "reward_meter_mean": 0.9968419075012207, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9968419075012207, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1052.0} {"timestamp_utc": "2026-04-11T21:39:53Z", "mode": "train", "global_step": 1053, "epoch": 0.04066265060240964, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.812121212121212e-06, "num_tokens": 2270753.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00017959998513106257, "sampling/sampling_logp_difference/max": 0.011735539883375168, "sampling/importance_sampling_ratio/min": 0.9999438524246216, "sampling/importance_sampling_ratio/mean": 1.0001797676086426, "sampling/importance_sampling_ratio/max": 1.0118045806884766, "entropy": 0.0012685270776273683, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1053.0} {"timestamp_utc": "2026-04-11T21:39:57Z", "mode": "train", "global_step": 1054, "epoch": 0.04070126660488106, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.80909090909091e-06, "num_tokens": 2272529.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00012025728210574016, "sampling/sampling_logp_difference/max": 0.008706126362085342, "sampling/importance_sampling_ratio/min": 0.9994588494300842, "sampling/importance_sampling_ratio/mean": 1.000115990638733, "sampling/importance_sampling_ratio/max": 1.008744239807129, "entropy": 0.0013022492494201288, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1054.0} {"timestamp_utc": "2026-04-11T21:40:02Z", "mode": "train", "global_step": 1055, "epoch": 0.040739882607352486, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.806060606060607e-06, "num_tokens": 2274175.0, "completions/mean_length": 50.75, "completions/min_length": 50.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9589456915855408, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9589456915855408, "rewards/total_composite/std": 0.0, "reward": 0.9589456915855408, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0029871375299990177, "sampling/sampling_logp_difference/max": 0.30701398849487305, "sampling/importance_sampling_ratio/min": 0.7356403470039368, "sampling/importance_sampling_ratio/mean": 1.0008618831634521, "sampling/importance_sampling_ratio/max": 1.2294468879699707, "entropy": 0.0185652831569314, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9589456915855408, "reward_meter_mean": 0.9589456915855408, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9589456915855408, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1055.0} {"timestamp_utc": "2026-04-11T21:40:07Z", "mode": "train", "global_step": 1056, "epoch": 0.04077849860982391, "loss": 0.008, "grad_norm": 3.9809470176696777, "learning_rate": 6.803030303030304e-06, "num_tokens": 2275998.0, "completions/mean_length": 62.875, "completions/min_length": 61.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.875, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.0021876515820622444, "rewards/meter/std": 0.0002445352729409933, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0021876515820622444, "rewards/total_composite/std": 0.0002445352729409933, "reward": 0.0021876515820622444, "reward_std": 0.00024453524383716285, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008529768325388432, "sampling/sampling_logp_difference/max": 1.1146979331970215, "sampling/importance_sampling_ratio/min": 0.6839931607246399, "sampling/importance_sampling_ratio/mean": 1.0037806034088135, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.041025768499821424, "clip_ratio/low_mean": 0.013767930213361979, "clip_ratio/low_min": 0.013767930213361979, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.015817110426723957, "reward_total_mean": 0.0021876515820622444, "reward_meter_mean": 0.0021876515820622444, "reward_meter_std": 0.0002445352729409933, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0021876515820622444, "reward_total_composite_std": 0.0002445352729409933, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1056.0} {"timestamp_utc": "2026-04-11T21:40:12Z", "mode": "train", "global_step": 1057, "epoch": 0.040817114612295334, "loss": -0.0561, "grad_norm": 0.9426341652870178, "learning_rate": 6.800000000000001e-06, "num_tokens": 2277958.0, "completions/mean_length": 83.0, "completions/min_length": 81.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.0, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9951313734054565, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7048847079277039, "rewards/total_composite/std": 0.1172773540019989, "reward": 0.7048847079277039, "reward_std": 0.1172773465514183, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00949889700859785, "sampling/sampling_logp_difference/max": 3.2940514087677, "sampling/importance_sampling_ratio/min": 0.03710322454571724, "sampling/importance_sampling_ratio/mean": 0.9974780082702637, "sampling/importance_sampling_ratio/max": 1.134204387664795, "entropy": 0.003993239166447893, "clip_ratio/low_mean": 0.0015432098880410194, "clip_ratio/low_min": 0.0015432098880410194, "clip_ratio/high_mean": 0.0012886597542092204, "clip_ratio/high_max": 0.0012886597542092204, "clip_ratio/region_mean": 0.00283186964225024, "reward_total_mean": 0.7048847079277039, "reward_meter_mean": 0.9951313734054565, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7048847079277039, "reward_total_composite_std": 0.1172773540019989, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1057.0} {"timestamp_utc": "2026-04-11T21:40:19Z", "mode": "train", "global_step": 1058, "epoch": 0.04085573061476676, "loss": 0.0039, "grad_norm": 0.21764229238033295, "learning_rate": 6.796969696969697e-06, "num_tokens": 2280914.0, "completions/mean_length": 198.5, "completions/min_length": 197.0, "completions/max_length": 209.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 198.5, "completions/min_terminated_length": 197.0, "completions/max_terminated_length": 209.0, "rewards/meter/mean": 0.9969083070755005, "rewards/meter/std": 3.72788890672382e-05, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7975266575813293, "rewards/total_composite/std": 2.9818897019140422e-05, "reward": 0.7975266575813293, "reward_std": 2.9827928301529028e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003239656798541546, "sampling/sampling_logp_difference/max": 3.510453224182129, "sampling/importance_sampling_ratio/min": 0.0298833679407835, "sampling/importance_sampling_ratio/mean": 0.9991796016693115, "sampling/importance_sampling_ratio/max": 1.167437195777893, "entropy": 0.0015565359972242732, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0006345177534967661, "clip_ratio/high_max": 0.0006345177534967661, "clip_ratio/region_mean": 0.0006345177534967661, "reward_total_mean": 0.7975266575813293, "reward_meter_mean": 0.9969083070755005, "reward_meter_std": 3.72788890672382e-05, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7975266575813293, "reward_total_composite_std": 2.9818897019140422e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1058.0} {"timestamp_utc": "2026-04-11T21:40:23Z", "mode": "train", "global_step": 1059, "epoch": 0.04089434661723818, "loss": 0.0234, "grad_norm": 4.7054972648620605, "learning_rate": 6.793939393939395e-06, "num_tokens": 2282612.0, "completions/mean_length": 54.25, "completions/min_length": 52.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.25, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9560332298278809, "rewards/meter/std": 0.0188444871455431, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9560332298278809, "rewards/total_composite/std": 0.0188444871455431, "reward": 0.9560332298278809, "reward_std": 0.018844490870833397, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008652369491755962, "sampling/sampling_logp_difference/max": 0.96638023853302, "sampling/importance_sampling_ratio/min": 0.38045772910118103, "sampling/importance_sampling_ratio/mean": 0.9992375373840332, "sampling/importance_sampling_ratio/max": 1.3655518293380737, "entropy": 0.025705090374685824, "clip_ratio/low_mean": 0.00657894741743803, "clip_ratio/low_min": 0.00657894741743803, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.00657894741743803, "reward_total_mean": 0.9560332298278809, "reward_meter_mean": 0.9560332298278809, "reward_meter_std": 0.0188444871455431, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9560332298278809, "reward_total_composite_std": 0.0188444871455431, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1059.0} {"timestamp_utc": "2026-04-11T21:40:31Z", "mode": "train", "global_step": 1060, "epoch": 0.040932962619709606, "loss": 0.0057, "grad_norm": 0.09687843173742294, "learning_rate": 6.790909090909091e-06, "num_tokens": 2286469.0, "completions/mean_length": 274.125, "completions/min_length": 272.0, "completions/max_length": 289.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 274.125, "completions/min_terminated_length": 272.0, "completions/max_terminated_length": 289.0, "rewards/meter/mean": 0.9969933032989502, "rewards/meter/std": 5.181956657906994e-05, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8723691701889038, "rewards/total_composite/std": 4.535001062322408e-05, "reward": 0.8723691701889038, "reward_std": 4.535001062322408e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00142471503932029, "sampling/sampling_logp_difference/max": 1.230623722076416, "sampling/importance_sampling_ratio/min": 0.2921103239059448, "sampling/importance_sampling_ratio/mean": 0.9990590214729309, "sampling/importance_sampling_ratio/max": 1.0306644439697266, "entropy": 0.0006548625879077008, "clip_ratio/low_mean": 0.00043252596515230834, "clip_ratio/low_min": 0.00043252596515230834, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/region_mean": 0.0022707612661179155, "reward_total_mean": 0.8723691701889038, "reward_meter_mean": 0.9969933032989502, "reward_meter_std": 5.181956657906994e-05, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8723691701889038, "reward_total_composite_std": 4.535001062322408e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1060.0} {"timestamp_utc": "2026-04-11T21:40:38Z", "mode": "train", "global_step": 1061, "epoch": 0.04097157862218103, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.787878787878789e-06, "num_tokens": 2290485.0, "completions/mean_length": 272.0, "completions/min_length": 272.0, "completions/max_length": 272.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 272.0, "completions/min_terminated_length": 272.0, "completions/max_terminated_length": 272.0, "rewards/meter/mean": 0.997011661529541, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8723852038383484, "rewards/total_composite/std": 0.0, "reward": 0.8723852038383484, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 6.0048791056033224e-05, "sampling/sampling_logp_difference/max": 0.03515136241912842, "sampling/importance_sampling_ratio/min": 0.9654592871665955, "sampling/importance_sampling_ratio/mean": 0.9999895691871643, "sampling/importance_sampling_ratio/max": 1.0054314136505127, "entropy": 0.00023714406961516943, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8723852038383484, "reward_meter_mean": 0.997011661529541, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8723852038383484, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1061.0} {"timestamp_utc": "2026-04-11T21:40:44Z", "mode": "train", "global_step": 1062, "epoch": 0.041010194624652455, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.7848484848484855e-06, "num_tokens": 2292821.0, "completions/mean_length": 106.0, "completions/min_length": 106.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.0, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.01337476447224617, "sampling/sampling_logp_difference/max": 7.350694179534912, "sampling/importance_sampling_ratio/min": 0.0006421464495360851, "sampling/importance_sampling_ratio/mean": 0.9978539347648621, "sampling/importance_sampling_ratio/max": 1.0233614444732666, "entropy": 0.0023513801133958623, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1062.0} {"timestamp_utc": "2026-04-11T21:40:48Z", "mode": "train", "global_step": 1063, "epoch": 0.04104881062712388, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.781818181818183e-06, "num_tokens": 2294669.0, "completions/mean_length": 65.0, "completions/min_length": 65.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9949578642845154, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9949578642845154, "rewards/total_composite/std": 0.0, "reward": 0.9949578642845154, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00023568868346046656, "sampling/sampling_logp_difference/max": 0.017239108681678772, "sampling/importance_sampling_ratio/min": 0.9829086661338806, "sampling/importance_sampling_ratio/mean": 1.0001155138015747, "sampling/importance_sampling_ratio/max": 1.011854648590088, "entropy": 0.0019514950545271859, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9949578642845154, "reward_meter_mean": 0.9949578642845154, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9949578642845154, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1063.0} {"timestamp_utc": "2026-04-11T21:40:53Z", "mode": "train", "global_step": 1064, "epoch": 0.0410874266295953, "loss": -0.0026, "grad_norm": 2.7271664142608643, "learning_rate": 6.778787878787879e-06, "num_tokens": 2296475.0, "completions/mean_length": 54.75, "completions/min_length": 53.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9718772172927856, "rewards/meter/std": 0.01329784281551838, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9718772172927856, "rewards/total_composite/std": 0.01329784281551838, "reward": 0.9718772172927856, "reward_std": 0.013297837227582932, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024222884327173233, "sampling/sampling_logp_difference/max": 5.190197944641113, "sampling/importance_sampling_ratio/min": 0.005570904351770878, "sampling/importance_sampling_ratio/mean": 0.9998058676719666, "sampling/importance_sampling_ratio/max": 1.5688531398773193, "entropy": 0.10237977746874094, "clip_ratio/low_mean": 0.009181267116218805, "clip_ratio/low_min": 0.009181267116218805, "clip_ratio/high_mean": 0.002314814832061529, "clip_ratio/high_max": 0.002314814832061529, "clip_ratio/region_mean": 0.011496081948280334, "reward_total_mean": 0.9718772172927856, "reward_meter_mean": 0.9718772172927856, "reward_meter_std": 0.01329784281551838, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9718772172927856, "reward_total_composite_std": 0.01329784281551838, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1064.0} {"timestamp_utc": "2026-04-11T21:40:58Z", "mode": "train", "global_step": 1065, "epoch": 0.04112604263206673, "loss": 0.0102, "grad_norm": 5.474052429199219, "learning_rate": 6.7757575757575765e-06, "num_tokens": 2298309.0, "completions/mean_length": 57.25, "completions/min_length": 57.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.25, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8288533687591553, "rewards/meter/std": 0.037421029061079025, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8288533687591553, "rewards/total_composite/std": 0.037421029061079025, "reward": 0.8288533687591553, "reward_std": 0.03742102161049843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002723454497754574, "sampling/sampling_logp_difference/max": 0.5241236686706543, "sampling/importance_sampling_ratio/min": 0.5920739769935608, "sampling/importance_sampling_ratio/mean": 0.9999019503593445, "sampling/importance_sampling_ratio/max": 1.1381598711013794, "entropy": 0.01174389524385333, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8288533687591553, "reward_meter_mean": 0.8288533687591553, "reward_meter_std": 0.037421029061079025, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8288533687591553, "reward_total_composite_std": 0.037421029061079025, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1065.0} {"timestamp_utc": "2026-04-11T21:41:02Z", "mode": "train", "global_step": 1066, "epoch": 0.04116465863453815, "loss": 0.0097, "grad_norm": 2.398463726043701, "learning_rate": 6.772727272727273e-06, "num_tokens": 2300092.0, "completions/mean_length": 60.875, "completions/min_length": 60.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9969611763954163, "rewards/meter/std": 0.0003373012295924127, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9969611763954163, "rewards/total_composite/std": 0.0003373012295924127, "reward": 0.9969611763954163, "reward_std": 0.00033730725408531725, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0037930156104266644, "sampling/sampling_logp_difference/max": 1.299863576889038, "sampling/importance_sampling_ratio/min": 0.2725689709186554, "sampling/importance_sampling_ratio/mean": 0.9994586110115051, "sampling/importance_sampling_ratio/max": 1.084521770477295, "entropy": 0.014361659123096615, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0020833334419876337, "clip_ratio/high_max": 0.0020833334419876337, "clip_ratio/region_mean": 0.0020833334419876337, "reward_total_mean": 0.9969611763954163, "reward_meter_mean": 0.9969611763954163, "reward_meter_std": 0.0003373012295924127, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9969611763954163, "reward_total_composite_std": 0.0003373012295924127, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1066.0} {"timestamp_utc": "2026-04-11T21:41:07Z", "mode": "train", "global_step": 1067, "epoch": 0.041203274637009575, "loss": 0.0034, "grad_norm": 10.166537284851074, "learning_rate": 6.76969696969697e-06, "num_tokens": 2301589.0, "completions/mean_length": 27.125, "completions/min_length": 26.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.125, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9321423768997192, "rewards/meter/std": 0.007329464890062809, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9321423768997192, "rewards/total_composite/std": 0.007329464890062809, "reward": 0.9321423768997192, "reward_std": 0.007329456973820925, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026033449918031693, "sampling/sampling_logp_difference/max": 1.5315618515014648, "sampling/importance_sampling_ratio/min": 0.21619774401187897, "sampling/importance_sampling_ratio/mean": 1.0001667737960815, "sampling/importance_sampling_ratio/max": 1.4033339023590088, "entropy": 0.09683473920449615, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.00893997447565198, "clip_ratio/high_max": 0.00893997447565198, "clip_ratio/region_mean": 0.00893997447565198, "reward_total_mean": 0.9321423768997192, "reward_meter_mean": 0.9321423768997192, "reward_meter_std": 0.007329464890062809, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9321423768997192, "reward_total_composite_std": 0.007329464890062809, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1067.0} {"timestamp_utc": "2026-04-11T21:41:12Z", "mode": "train", "global_step": 1068, "epoch": 0.041241890639481, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.7666666666666665e-06, "num_tokens": 2303405.0, "completions/mean_length": 65.0, "completions/min_length": 65.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9949578642845154, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9949578642845154, "rewards/total_composite/std": 0.0, "reward": 0.9949578642845154, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00019105577666778117, "sampling/sampling_logp_difference/max": 0.0064575872384011745, "sampling/importance_sampling_ratio/min": 0.9977074861526489, "sampling/importance_sampling_ratio/mean": 1.000160813331604, "sampling/importance_sampling_ratio/max": 1.0064785480499268, "entropy": 0.0018052591913146898, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9949578642845154, "reward_meter_mean": 0.9949578642845154, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9949578642845154, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1068.0} {"timestamp_utc": "2026-04-11T21:41:17Z", "mode": "train", "global_step": 1069, "epoch": 0.04128050664195242, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.763636363636365e-06, "num_tokens": 2305133.0, "completions/mean_length": 65.0, "completions/min_length": 65.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9949578642845154, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9949578642845154, "rewards/total_composite/std": 0.0, "reward": 0.9949578642845154, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005417983047664165, "sampling/sampling_logp_difference/max": 0.036248765885829926, "sampling/importance_sampling_ratio/min": 0.9796718955039978, "sampling/importance_sampling_ratio/mean": 1.0003167390823364, "sampling/importance_sampling_ratio/max": 1.0369137525558472, "entropy": 0.005523053434444591, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9949578642845154, "reward_meter_mean": 0.9949578642845154, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9949578642845154, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1069.0} {"timestamp_utc": "2026-04-11T21:41:21Z", "mode": "train", "global_step": 1070, "epoch": 0.04131912264442385, "loss": 0.0381, "grad_norm": 26.325374603271484, "learning_rate": 6.760606060606061e-06, "num_tokens": 2306762.0, "completions/mean_length": 52.625, "completions/min_length": 50.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.11274315416812897, "rewards/meter/std": 0.3188244700431824, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.11274315416812897, "rewards/total_composite/std": 0.3188244700431824, "reward": 0.11274315416812897, "reward_std": 0.31882444024086, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02075301483273506, "sampling/sampling_logp_difference/max": 1.0765538215637207, "sampling/importance_sampling_ratio/min": 0.34076783061027527, "sampling/importance_sampling_ratio/mean": 1.0003687143325806, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05561392899835482, "clip_ratio/low_mean": 0.016467438312247396, "clip_ratio/low_min": 0.016467438312247396, "clip_ratio/high_mean": 0.007499999832361937, "clip_ratio/high_max": 0.007499999832361937, "clip_ratio/region_mean": 0.023967438144609332, "reward_total_mean": 0.11274315416812897, "reward_meter_mean": 0.11274315416812897, "reward_meter_std": 0.3188244700431824, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.11274315416812897, "reward_total_composite_std": 0.3188244700431824, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1070.0} {"timestamp_utc": "2026-04-11T21:41:26Z", "mode": "train", "global_step": 1071, "epoch": 0.04135773864689527, "loss": -0.0217, "grad_norm": 5.607251167297363, "learning_rate": 6.757575757575758e-06, "num_tokens": 2308113.0, "completions/mean_length": 25.875, "completions/min_length": 25.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.875, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9891167879104614, "rewards/meter/std": 0.008078054524958134, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9891167879104614, "rewards/total_composite/std": 0.008078054524958134, "reward": 0.9891167879104614, "reward_std": 0.008078045211732388, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009091660380363464, "sampling/sampling_logp_difference/max": 0.6194605827331543, "sampling/importance_sampling_ratio/min": 0.5382347106933594, "sampling/importance_sampling_ratio/mean": 0.9988547563552856, "sampling/importance_sampling_ratio/max": 1.0410070419311523, "entropy": 0.02826369390822947, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9891167879104614, "reward_meter_mean": 0.9891167879104614, "reward_meter_std": 0.008078054524958134, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9891167879104614, "reward_total_composite_std": 0.008078054524958134, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1071.0} {"timestamp_utc": "2026-04-11T21:41:31Z", "mode": "train", "global_step": 1072, "epoch": 0.041396354649366696, "loss": 0.0001, "grad_norm": 2.4955992698669434, "learning_rate": 6.754545454545455e-06, "num_tokens": 2309830.0, "completions/mean_length": 61.625, "completions/min_length": 61.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.997157096862793, "rewards/meter/std": 0.0005752869765274227, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.997157096862793, "rewards/total_composite/std": 0.0005752869765274227, "reward": 0.997157096862793, "reward_std": 0.000575280690100044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012418783269822598, "sampling/sampling_logp_difference/max": 0.5172317028045654, "sampling/importance_sampling_ratio/min": 0.5961686372756958, "sampling/importance_sampling_ratio/mean": 1.0032328367233276, "sampling/importance_sampling_ratio/max": 1.3372715711593628, "entropy": 0.10848722152877599, "clip_ratio/low_mean": 0.010146747343242168, "clip_ratio/low_min": 0.010146747343242168, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.010146747343242168, "reward_total_mean": 0.997157096862793, "reward_meter_mean": 0.997157096862793, "reward_meter_std": 0.0005752869765274227, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.997157096862793, "reward_total_composite_std": 0.0005752869765274227, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1072.0} {"timestamp_utc": "2026-04-11T21:41:36Z", "mode": "train", "global_step": 1073, "epoch": 0.04143497065183812, "loss": 0.0115, "grad_norm": 0.9153868556022644, "learning_rate": 6.751515151515152e-06, "num_tokens": 2311287.0, "completions/mean_length": 33.125, "completions/min_length": 33.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9944090247154236, "rewards/meter/std": 0.0004998616641387343, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9944090247154236, "rewards/total_composite/std": 0.0004998616641387343, "reward": 0.9944090247154236, "reward_std": 0.0004998616059310734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005571494810283184, "sampling/sampling_logp_difference/max": 0.6732907295227051, "sampling/importance_sampling_ratio/min": 0.9340354800224304, "sampling/importance_sampling_ratio/mean": 1.0061020851135254, "sampling/importance_sampling_ratio/max": 1.9606788158416748, "entropy": 0.025314107653684914, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0036764706019312143, "reward_total_mean": 0.9944090247154236, "reward_meter_mean": 0.9944090247154236, "reward_meter_std": 0.0004998616641387343, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9944090247154236, "reward_total_composite_std": 0.0004998616641387343, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1073.0} {"timestamp_utc": "2026-04-11T21:41:41Z", "mode": "train", "global_step": 1074, "epoch": 0.041473586654309544, "loss": 0.0417, "grad_norm": 3.6375036239624023, "learning_rate": 6.748484848484848e-06, "num_tokens": 2313096.0, "completions/mean_length": 54.125, "completions/min_length": 51.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9875502586364746, "rewards/meter/std": 0.003662221832200885, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9875502586364746, "rewards/total_composite/std": 0.003662221832200885, "reward": 0.9875502586364746, "reward_std": 0.003662231843918562, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004612638149410486, "sampling/sampling_logp_difference/max": 0.22617030143737793, "sampling/importance_sampling_ratio/min": 0.7975822687149048, "sampling/importance_sampling_ratio/mean": 1.003567099571228, "sampling/importance_sampling_ratio/max": 1.2383108139038086, "entropy": 0.03697874676436186, "clip_ratio/low_mean": 0.004464285913854837, "clip_ratio/low_min": 0.004464285913854837, "clip_ratio/high_mean": 0.0024509804788976908, "clip_ratio/high_max": 0.0024509804788976908, "clip_ratio/region_mean": 0.006915266392752528, "reward_total_mean": 0.9875502586364746, "reward_meter_mean": 0.9875502586364746, "reward_meter_std": 0.003662221832200885, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9875502586364746, "reward_total_composite_std": 0.003662221832200885, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1074.0} {"timestamp_utc": "2026-04-11T21:41:46Z", "mode": "train", "global_step": 1075, "epoch": 0.04151220265678097, "loss": 0.0288, "grad_norm": 6.4112677574157715, "learning_rate": 6.7454545454545465e-06, "num_tokens": 2314778.0, "completions/mean_length": 47.25, "completions/min_length": 41.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.10002761334180832, "rewards/meter/std": 0.26244989037513733, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.10002761334180832, "rewards/total_composite/std": 0.26244989037513733, "reward": 0.10002761334180832, "reward_std": 0.26244989037513733, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03375272452831268, "sampling/sampling_logp_difference/max": 1.2938915491104126, "sampling/importance_sampling_ratio/min": 0.2742016315460205, "sampling/importance_sampling_ratio/mean": 1.0019086599349976, "sampling/importance_sampling_ratio/max": 1.4211242198944092, "entropy": 0.16682779975235462, "clip_ratio/low_mean": 0.03481511096470058, "clip_ratio/low_min": 0.03481511096470058, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.03481511096470058, "reward_total_mean": 0.10002761334180832, "reward_meter_mean": 0.10002761334180832, "reward_meter_std": 0.26244989037513733, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.10002761334180832, "reward_total_composite_std": 0.26244989037513733, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1075.0} {"timestamp_utc": "2026-04-11T21:41:51Z", "mode": "train", "global_step": 1076, "epoch": 0.04155081865925239, "loss": -0.0001, "grad_norm": 4.325499057769775, "learning_rate": 6.742424242424243e-06, "num_tokens": 2316531.0, "completions/mean_length": 53.125, "completions/min_length": 52.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9805549383163452, "rewards/meter/std": 0.00861036404967308, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9805549383163452, "rewards/total_composite/std": 0.00861036404967308, "reward": 0.9805549383163452, "reward_std": 0.008610363118350506, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011005771346390247, "sampling/sampling_logp_difference/max": 0.7941849231719971, "sampling/importance_sampling_ratio/min": 0.45194947719573975, "sampling/importance_sampling_ratio/mean": 1.0017882585525513, "sampling/importance_sampling_ratio/max": 1.5676127672195435, "entropy": 0.049260836094617844, "clip_ratio/low_mean": 0.009479318046942353, "clip_ratio/low_min": 0.009479318046942353, "clip_ratio/high_mean": 0.004631217801943421, "clip_ratio/high_max": 0.004631217801943421, "clip_ratio/region_mean": 0.014110535848885775, "reward_total_mean": 0.9805549383163452, "reward_meter_mean": 0.9805549383163452, "reward_meter_std": 0.00861036404967308, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9805549383163452, "reward_total_composite_std": 0.00861036404967308, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1076.0} {"timestamp_utc": "2026-04-11T21:41:56Z", "mode": "train", "global_step": 1077, "epoch": 0.041589434661723816, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.73939393939394e-06, "num_tokens": 2318435.0, "completions/mean_length": 65.0, "completions/min_length": 65.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9949578642845154, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9949578642845154, "rewards/total_composite/std": 0.0, "reward": 0.9949578642845154, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.000207486460567452, "sampling/sampling_logp_difference/max": 0.02866499498486519, "sampling/importance_sampling_ratio/min": 0.9717419147491455, "sampling/importance_sampling_ratio/mean": 1.000050663948059, "sampling/importance_sampling_ratio/max": 1.0065354108810425, "entropy": 0.0024144309863913804, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9949578642845154, "reward_meter_mean": 0.9949578642845154, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9949578642845154, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1077.0} {"timestamp_utc": "2026-04-11T21:42:01Z", "mode": "train", "global_step": 1078, "epoch": 0.04162805066419524, "loss": 0.0161, "grad_norm": 4.65764856338501, "learning_rate": 6.7363636363636365e-06, "num_tokens": 2320094.0, "completions/mean_length": 61.375, "completions/min_length": 59.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9684683084487915, "rewards/meter/std": 0.08065449446439743, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9684683084487915, "rewards/total_composite/std": 0.08065449446439743, "reward": 0.9684683084487915, "reward_std": 0.08065447211265564, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02014295756816864, "sampling/sampling_logp_difference/max": 1.577573299407959, "sampling/importance_sampling_ratio/min": 0.20647554099559784, "sampling/importance_sampling_ratio/mean": 1.0008001327514648, "sampling/importance_sampling_ratio/max": 1.5210373401641846, "entropy": 0.13324717595241964, "clip_ratio/low_mean": 0.0078125, "clip_ratio/low_min": 0.0078125, "clip_ratio/high_mean": 0.010216211201623082, "clip_ratio/high_max": 0.010216211201623082, "clip_ratio/region_mean": 0.018028711201623082, "reward_total_mean": 0.9684683084487915, "reward_meter_mean": 0.9684683084487915, "reward_meter_std": 0.08065449446439743, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9684683084487915, "reward_total_composite_std": 0.08065449446439743, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1078.0} {"timestamp_utc": "2026-04-11T21:42:06Z", "mode": "train", "global_step": 1079, "epoch": 0.041666666666666664, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.733333333333334e-06, "num_tokens": 2321786.0, "completions/mean_length": 49.5, "completions/min_length": 46.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.5, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.091847725212574, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.091847725212574, "rewards/total_composite/std": 0.0, "reward": 0.091847725212574, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.01916210725903511, "sampling/sampling_logp_difference/max": 1.860931396484375, "sampling/importance_sampling_ratio/min": 0.15552771091461182, "sampling/importance_sampling_ratio/mean": 0.9989585280418396, "sampling/importance_sampling_ratio/max": 1.5695313215255737, "entropy": 0.054232243448495865, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.091847725212574, "reward_meter_mean": 0.091847725212574, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.091847725212574, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1079.0} {"timestamp_utc": "2026-04-11T21:42:11Z", "mode": "train", "global_step": 1080, "epoch": 0.04170528266913809, "loss": 0.0236, "grad_norm": 1.9305036067962646, "learning_rate": 6.73030303030303e-06, "num_tokens": 2324566.0, "completions/mean_length": 132.5, "completions/min_length": 129.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.5, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9948593974113464, "rewards/meter/std": 0.002004272071644664, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.663239598274231, "rewards/total_composite/std": 0.0013361814199015498, "reward": 0.663239598274231, "reward_std": 0.0013361814199015498, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004363351967185736, "sampling/sampling_logp_difference/max": 0.48967456817626953, "sampling/importance_sampling_ratio/min": 0.6128258109092712, "sampling/importance_sampling_ratio/mean": 1.0016835927963257, "sampling/importance_sampling_ratio/max": 1.3798326253890991, "entropy": 0.029038164531812072, "clip_ratio/low_mean": 0.0027643622015602887, "clip_ratio/low_min": 0.0027643622015602887, "clip_ratio/high_mean": 0.0019379844889044762, "clip_ratio/high_max": 0.0019379844889044762, "clip_ratio/region_mean": 0.004702346690464765, "reward_total_mean": 0.663239598274231, "reward_meter_mean": 0.9948593974113464, "reward_meter_std": 0.002004272071644664, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.663239598274231, "reward_total_composite_std": 0.0013361814199015498, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1080.0} {"timestamp_utc": "2026-04-11T21:42:17Z", "mode": "train", "global_step": 1081, "epoch": 0.04174389867160951, "loss": 0.0025, "grad_norm": 0.31140798330307007, "learning_rate": 6.7272727272727275e-06, "num_tokens": 2327382.0, "completions/mean_length": 175.0, "completions/min_length": 161.0, "completions/max_length": 178.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 175.0, "completions/min_terminated_length": 161.0, "completions/max_terminated_length": 178.0, "rewards/meter/mean": 0.9966549277305603, "rewards/meter/std": 7.61248666094616e-05, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.747491180896759, "rewards/total_composite/std": 5.710634286515415e-05, "reward": 0.747491180896759, "reward_std": 5.710850018658675e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0060366359539330006, "sampling/sampling_logp_difference/max": 3.635336399078369, "sampling/importance_sampling_ratio/min": 0.0263750609010458, "sampling/importance_sampling_ratio/mean": 0.9993287920951843, "sampling/importance_sampling_ratio/max": 1.509435772895813, "entropy": 0.015698739560320973, "clip_ratio/low_mean": 0.001416441984474659, "clip_ratio/low_min": 0.001416441984474659, "clip_ratio/high_mean": 0.002259009750559926, "clip_ratio/high_max": 0.002259009750559926, "clip_ratio/region_mean": 0.003675451735034585, "reward_total_mean": 0.747491180896759, "reward_meter_mean": 0.9966549277305603, "reward_meter_std": 7.61248666094616e-05, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.747491180896759, "reward_total_composite_std": 5.710634286515415e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1081.0} {"timestamp_utc": "2026-04-11T21:42:23Z", "mode": "train", "global_step": 1082, "epoch": 0.04178251467408094, "loss": 0.0114, "grad_norm": 4.485145092010498, "learning_rate": 6.724242424242424e-06, "num_tokens": 2329495.0, "completions/mean_length": 108.125, "completions/min_length": 106.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.125, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9212300777435303, "rewards/meter/std": 0.21582916378974915, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9212300777435303, "rewards/total_composite/std": 0.21582916378974915, "reward": 0.9212300777435303, "reward_std": 0.21582916378974915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013148986734449863, "sampling/sampling_logp_difference/max": 0.9551130533218384, "sampling/importance_sampling_ratio/min": 0.38476866483688354, "sampling/importance_sampling_ratio/mean": 1.0012389421463013, "sampling/importance_sampling_ratio/max": 1.5335668325424194, "entropy": 0.09740556543692946, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.009324767161160707, "clip_ratio/high_max": 0.009324767161160707, "clip_ratio/region_mean": 0.012703145621344447, "reward_total_mean": 0.9212300777435303, "reward_meter_mean": 0.9212300777435303, "reward_meter_std": 0.21582916378974915, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9212300777435303, "reward_total_composite_std": 0.21582916378974915, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1082.0} {"timestamp_utc": "2026-04-11T21:42:28Z", "mode": "train", "global_step": 1083, "epoch": 0.04182113067655236, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.721212121212122e-06, "num_tokens": 2331783.0, "completions/mean_length": 129.0, "completions/min_length": 129.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.0, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9951313734054565, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6634209156036377, "rewards/total_composite/std": 0.0, "reward": 0.6634209156036377, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002578256244305521, "sampling/sampling_logp_difference/max": 0.045966774225234985, "sampling/importance_sampling_ratio/min": 0.9550737738609314, "sampling/importance_sampling_ratio/mean": 1.0000473260879517, "sampling/importance_sampling_ratio/max": 1.0351933240890503, "entropy": 0.002667208347702399, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.6634209156036377, "reward_meter_mean": 0.9951313734054565, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6634209156036377, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1083.0} {"timestamp_utc": "2026-04-11T21:42:33Z", "mode": "train", "global_step": 1084, "epoch": 0.041859746679023785, "loss": 0.0462, "grad_norm": 17.376197814941406, "learning_rate": 6.718181818181819e-06, "num_tokens": 2333323.0, "completions/mean_length": 28.5, "completions/min_length": 27.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.6803448796272278, "rewards/meter/std": 0.4452557861804962, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6803448796272278, "rewards/total_composite/std": 0.4452557861804962, "reward": 0.6803448796272278, "reward_std": 0.4452557861804962, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07098246365785599, "sampling/sampling_logp_difference/max": 2.6754026412963867, "sampling/importance_sampling_ratio/min": 0.06887909024953842, "sampling/importance_sampling_ratio/mean": 0.9909398555755615, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25689972564578056, "clip_ratio/low_mean": 0.012652947567403316, "clip_ratio/low_min": 0.012652947567403316, "clip_ratio/high_mean": 0.04814814869314432, "clip_ratio/high_max": 0.04814814869314432, "clip_ratio/region_mean": 0.06080109626054764, "reward_total_mean": 0.6803448796272278, "reward_meter_mean": 0.6803448796272278, "reward_meter_std": 0.4452557861804962, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6803448796272278, "reward_total_composite_std": 0.4452557861804962, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1084.0} {"timestamp_utc": "2026-04-11T21:42:40Z", "mode": "train", "global_step": 1085, "epoch": 0.04189836268149521, "loss": -0.0223, "grad_norm": 7.492648124694824, "learning_rate": 6.715151515151516e-06, "num_tokens": 2336298.0, "completions/mean_length": 189.875, "completions/min_length": 180.0, "completions/max_length": 201.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 189.875, "completions/min_terminated_length": 180.0, "completions/max_terminated_length": 201.0, "rewards/meter/mean": 0.46110403537750244, "rewards/meter/std": 0.25172099471092224, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.07715165615081787, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3282319903373718, "rewards/total_composite/std": 0.1757289618253708, "reward": 0.3282319903373718, "reward_std": 0.1757289618253708, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010337450541555882, "sampling/sampling_logp_difference/max": 1.9266459941864014, "sampling/importance_sampling_ratio/min": 0.14563584327697754, "sampling/importance_sampling_ratio/mean": 1.0021684169769287, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.034014943055808544, "clip_ratio/low_mean": 0.002667596214450896, "clip_ratio/low_min": 0.002667596214450896, "clip_ratio/high_mean": 0.0026115492219105363, "clip_ratio/high_max": 0.0026115492219105363, "clip_ratio/region_mean": 0.005279145436361432, "reward_total_mean": 0.3282319903373718, "reward_meter_mean": 0.46110403537750244, "reward_meter_std": 0.25172099471092224, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.07715165615081787, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3282319903373718, "reward_total_composite_std": 0.1757289618253708, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1085.0} {"timestamp_utc": "2026-04-11T21:42:46Z", "mode": "train", "global_step": 1086, "epoch": 0.04193697868396663, "loss": 0.0308, "grad_norm": 3.549293279647827, "learning_rate": 6.712121212121213e-06, "num_tokens": 2339268.0, "completions/mean_length": 208.25, "completions/min_length": 197.0, "completions/max_length": 213.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 208.25, "completions/min_terminated_length": 197.0, "completions/max_terminated_length": 213.0, "rewards/meter/mean": 0.9969711303710938, "rewards/meter/std": 8.204347250284627e-05, "rewards/count_adherence/mean": 0.6500000357627869, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6480287909507751, "rewards/total_composite/std": 0.0922786295413971, "reward": 0.6480287909507751, "reward_std": 0.09227863699197769, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010134638287127018, "sampling/sampling_logp_difference/max": 8.085393905639648, "sampling/importance_sampling_ratio/min": 0.00030800519743934274, "sampling/importance_sampling_ratio/mean": 1.0001766681671143, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.008055432001128793, "clip_ratio/low_mean": 0.002358543104492128, "clip_ratio/low_min": 0.002358543104492128, "clip_ratio/high_mean": 0.0006345177534967661, "clip_ratio/high_max": 0.0006345177534967661, "clip_ratio/region_mean": 0.002993060857988894, "reward_total_mean": 0.6480287909507751, "reward_meter_mean": 0.9969711303710938, "reward_meter_std": 8.204347250284627e-05, "reward_count_adherence_mean": 0.6500000357627869, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6480287909507751, "reward_total_composite_std": 0.0922786295413971, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1086.0} {"timestamp_utc": "2026-04-11T21:42:51Z", "mode": "train", "global_step": 1087, "epoch": 0.04197559468643806, "loss": -0.0111, "grad_norm": 2.269176959991455, "learning_rate": 6.709090909090909e-06, "num_tokens": 2341095.0, "completions/mean_length": 80.375, "completions/min_length": 76.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.375, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9966411590576172, "rewards/meter/std": 0.00015710237494204193, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9966411590576172, "rewards/total_composite/std": 0.00015710237494204193, "reward": 0.9966411590576172, "reward_std": 0.00015711141168139875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0034698506351560354, "sampling/sampling_logp_difference/max": 0.793515682220459, "sampling/importance_sampling_ratio/min": 0.45225203037261963, "sampling/importance_sampling_ratio/mean": 0.9995546340942383, "sampling/importance_sampling_ratio/max": 1.1145241260528564, "entropy": 0.014752325252629817, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0015432098880410194, "clip_ratio/high_max": 0.0015432098880410194, "clip_ratio/region_mean": 0.0015432098880410194, "reward_total_mean": 0.9966411590576172, "reward_meter_mean": 0.9966411590576172, "reward_meter_std": 0.00015710237494204193, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9966411590576172, "reward_total_composite_std": 0.00015710237494204193, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1087.0} {"timestamp_utc": "2026-04-11T21:42:57Z", "mode": "train", "global_step": 1088, "epoch": 0.04201421068890948, "loss": 0.0123, "grad_norm": 12.239304542541504, "learning_rate": 6.706060606060607e-06, "num_tokens": 2342713.0, "completions/mean_length": 49.25, "completions/min_length": 49.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.25, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.986293613910675, "rewards/meter/std": 0.0026903701946139336, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.986293613910675, "rewards/total_composite/std": 0.0026903701946139336, "reward": 0.986293613910675, "reward_std": 0.0026903690304607153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016510505229234695, "sampling/sampling_logp_difference/max": 1.4381656646728516, "sampling/importance_sampling_ratio/min": 0.23736277222633362, "sampling/importance_sampling_ratio/mean": 1.0027153491973877, "sampling/importance_sampling_ratio/max": 1.9696929454803467, "entropy": 0.09569864999502897, "clip_ratio/low_mean": 0.007452981313690543, "clip_ratio/low_min": 0.007452981313690543, "clip_ratio/high_mean": 0.007653061067685485, "clip_ratio/high_max": 0.007653061067685485, "clip_ratio/region_mean": 0.015106042381376028, "reward_total_mean": 0.986293613910675, "reward_meter_mean": 0.986293613910675, "reward_meter_std": 0.0026903701946139336, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.986293613910675, "reward_total_composite_std": 0.0026903701946139336, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1088.0} {"timestamp_utc": "2026-04-11T21:43:03Z", "mode": "train", "global_step": 1089, "epoch": 0.042052826691380905, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.703030303030304e-06, "num_tokens": 2345457.0, "completions/mean_length": 152.0, "completions/min_length": 152.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.0, "completions/min_terminated_length": 152.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.9968674182891846, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7476505637168884, "rewards/total_composite/std": 0.0, "reward": 0.7476505637168884, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006623952649533749, "sampling/sampling_logp_difference/max": 0.3456934988498688, "sampling/importance_sampling_ratio/min": 0.7077293992042542, "sampling/importance_sampling_ratio/mean": 0.9997900128364563, "sampling/importance_sampling_ratio/max": 1.08553946018219, "entropy": 0.0034695666399784386, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7476505637168884, "reward_meter_mean": 0.9968674182891846, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7476505637168884, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1089.0} {"timestamp_utc": "2026-04-11T21:43:09Z", "mode": "train", "global_step": 1090, "epoch": 0.04209144269385233, "loss": -0.0003, "grad_norm": 0.4215506911277771, "learning_rate": 6.700000000000001e-06, "num_tokens": 2348309.0, "completions/mean_length": 182.5, "completions/min_length": 181.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 182.5, "completions/min_terminated_length": 181.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9972051978111267, "rewards/meter/std": 5.313828296493739e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9972051978111267, "rewards/total_composite/std": 5.313828296493739e-05, "reward": 0.9972051978111267, "reward_std": 5.313552901498042e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004990379326045513, "sampling/sampling_logp_difference/max": 0.824894905090332, "sampling/importance_sampling_ratio/min": 0.4382810592651367, "sampling/importance_sampling_ratio/mean": 1.0013642311096191, "sampling/importance_sampling_ratio/max": 1.4445006847381592, "entropy": 0.04286058805882931, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004788926860783249, "clip_ratio/high_max": 0.004788926860783249, "clip_ratio/region_mean": 0.004788926860783249, "reward_total_mean": 0.9972051978111267, "reward_meter_mean": 0.9972051978111267, "reward_meter_std": 5.313828296493739e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9972051978111267, "reward_total_composite_std": 5.313828296493739e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1090.0} {"timestamp_utc": "2026-04-11T21:43:14Z", "mode": "train", "global_step": 1091, "epoch": 0.042130058696323754, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.6969696969696975e-06, "num_tokens": 2349895.0, "completions/mean_length": 47.25, "completions/min_length": 46.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.091847725212574, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.091847725212574, "rewards/total_composite/std": 0.0, "reward": 0.091847725212574, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.007502844091504812, "sampling/sampling_logp_difference/max": 0.6710173487663269, "sampling/importance_sampling_ratio/min": 0.511188268661499, "sampling/importance_sampling_ratio/mean": 1.001801609992981, "sampling/importance_sampling_ratio/max": 1.5886342525482178, "entropy": 0.016990114585496485, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.091847725212574, "reward_meter_mean": 0.091847725212574, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.091847725212574, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1091.0} {"timestamp_utc": "2026-04-11T21:43:18Z", "mode": "train", "global_step": 1092, "epoch": 0.04216867469879518, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.693939393939395e-06, "num_tokens": 2351375.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9955702424049377, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9955702424049377, "rewards/total_composite/std": 0.0, "reward": 0.9955702424049377, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.001951782964169979, "sampling/sampling_logp_difference/max": 0.17160826921463013, "sampling/importance_sampling_ratio/min": 0.8423091173171997, "sampling/importance_sampling_ratio/mean": 0.9996309876441956, "sampling/importance_sampling_ratio/max": 1.054145336151123, "entropy": 0.010097296675667167, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9955702424049377, "reward_meter_mean": 0.9955702424049377, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9955702424049377, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1092.0} {"timestamp_utc": "2026-04-11T21:43:23Z", "mode": "train", "global_step": 1093, "epoch": 0.0422072907012666, "loss": 0.0095, "grad_norm": 1.6124061346054077, "learning_rate": 6.690909090909091e-06, "num_tokens": 2353053.0, "completions/mean_length": 64.75, "completions/min_length": 63.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.75, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9950487613677979, "rewards/meter/std": 0.0002570115029811859, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9950487613677979, "rewards/total_composite/std": 0.0002570115029811859, "reward": 0.9950487613677979, "reward_std": 0.00025701147387735546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005157488863915205, "sampling/sampling_logp_difference/max": 1.5352113246917725, "sampling/importance_sampling_ratio/min": 0.21541017293930054, "sampling/importance_sampling_ratio/mean": 0.9983689785003662, "sampling/importance_sampling_ratio/max": 1.0859289169311523, "entropy": 0.013940075354184955, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.003846153849735856, "reward_total_mean": 0.9950487613677979, "reward_meter_mean": 0.9950487613677979, "reward_meter_std": 0.0002570115029811859, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9950487613677979, "reward_total_composite_std": 0.0002570115029811859, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1093.0} {"timestamp_utc": "2026-04-11T21:43:28Z", "mode": "train", "global_step": 1094, "epoch": 0.042245906703738026, "loss": 0.0094, "grad_norm": 3.3892717361450195, "learning_rate": 6.687878787878788e-06, "num_tokens": 2354894.0, "completions/mean_length": 65.125, "completions/min_length": 64.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9951438903808594, "rewards/meter/std": 0.0001731093943817541, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9951438903808594, "rewards/total_composite/std": 0.0001731093943817541, "reward": 0.9951438903808594, "reward_std": 0.00017311105330009013, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011436189524829388, "sampling/sampling_logp_difference/max": 1.639451026916504, "sampling/importance_sampling_ratio/min": 0.19408656656742096, "sampling/importance_sampling_ratio/mean": 0.996876060962677, "sampling/importance_sampling_ratio/max": 1.232495665550232, "entropy": 0.04774620302487165, "clip_ratio/low_mean": 0.0038470644503831863, "clip_ratio/low_min": 0.0038470644503831863, "clip_ratio/high_mean": 0.00390625, "clip_ratio/high_max": 0.00390625, "clip_ratio/region_mean": 0.007753314450383186, "reward_total_mean": 0.9951438903808594, "reward_meter_mean": 0.9951438903808594, "reward_meter_std": 0.0001731093943817541, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9951438903808594, "reward_total_composite_std": 0.0001731093943817541, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1094.0} {"timestamp_utc": "2026-04-11T21:43:33Z", "mode": "train", "global_step": 1095, "epoch": 0.04228452270620945, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.684848484848485e-06, "num_tokens": 2357158.0, "completions/mean_length": 107.0, "completions/min_length": 107.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.0, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9966511130332947, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9966511130332947, "rewards/total_composite/std": 0.0, "reward": 0.9966511130332947, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0007139050285331905, "sampling/sampling_logp_difference/max": 0.2203187644481659, "sampling/importance_sampling_ratio/min": 0.8022630214691162, "sampling/importance_sampling_ratio/mean": 0.999952495098114, "sampling/importance_sampling_ratio/max": 1.083788275718689, "entropy": 0.003691258083563298, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9966511130332947, "reward_meter_mean": 0.9966511130332947, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9966511130332947, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1095.0} {"timestamp_utc": "2026-04-11T21:43:38Z", "mode": "train", "global_step": 1096, "epoch": 0.042323138708680874, "loss": 0.0528, "grad_norm": 17.096412658691406, "learning_rate": 6.681818181818183e-06, "num_tokens": 2359061.0, "completions/mean_length": 26.875, "completions/min_length": 20.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.875, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.11241252720355988, "rewards/meter/std": 0.3034849762916565, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.11241252720355988, "rewards/total_composite/std": 0.3034849762916565, "reward": 0.11241252720355988, "reward_std": 0.3034849464893341, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10703544318675995, "sampling/sampling_logp_difference/max": 4.594521999359131, "sampling/importance_sampling_ratio/min": 0.010107050649821758, "sampling/importance_sampling_ratio/mean": 0.979833722114563, "sampling/importance_sampling_ratio/max": 1.9887901544570923, "entropy": 0.21645524073392153, "clip_ratio/low_mean": 0.03616698645055294, "clip_ratio/low_min": 0.03616698645055294, "clip_ratio/high_mean": 0.0223214291036129, "clip_ratio/high_max": 0.0223214291036129, "clip_ratio/region_mean": 0.05848841555416584, "reward_total_mean": 0.11241252720355988, "reward_meter_mean": 0.11241252720355988, "reward_meter_std": 0.3034849762916565, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.11241252720355988, "reward_total_composite_std": 0.3034849762916565, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1096.0} {"timestamp_utc": "2026-04-11T21:43:44Z", "mode": "train", "global_step": 1097, "epoch": 0.0423617547111523, "loss": 0.0354, "grad_norm": 0.9273874163627625, "learning_rate": 6.678787878787879e-06, "num_tokens": 2361662.0, "completions/mean_length": 147.125, "completions/min_length": 145.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.125, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.9964081048965454, "rewards/meter/std": 0.00033037204411812127, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9652670621871948, "rewards/total_composite/std": 0.08803740888834, "reward": 0.9652670621871948, "reward_std": 0.08803740888834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0055220527574419975, "sampling/sampling_logp_difference/max": 1.9974420070648193, "sampling/importance_sampling_ratio/min": 0.4937727451324463, "sampling/importance_sampling_ratio/mean": 1.0020689964294434, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.016805213410407305, "clip_ratio/low_mean": 0.000776397529989481, "clip_ratio/low_min": 0.000776397529989481, "clip_ratio/high_mean": 0.0025862068869173527, "clip_ratio/high_max": 0.0025862068869173527, "clip_ratio/region_mean": 0.0033626044169068336, "reward_total_mean": 0.9652670621871948, "reward_meter_mean": 0.9964081048965454, "reward_meter_std": 0.00033037204411812127, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9652670621871948, "reward_total_composite_std": 0.08803740888834, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1097.0} {"timestamp_utc": "2026-04-11T21:43:50Z", "mode": "train", "global_step": 1098, "epoch": 0.04240037071362372, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.6757575757575766e-06, "num_tokens": 2364710.0, "completions/mean_length": 167.0, "completions/min_length": 167.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 167.0, "completions/min_terminated_length": 167.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.9968674182891846, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8307228684425354, "rewards/total_composite/std": 0.0, "reward": 0.8307228684425354, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00022596628696192056, "sampling/sampling_logp_difference/max": 0.1071237251162529, "sampling/importance_sampling_ratio/min": 0.8984145522117615, "sampling/importance_sampling_ratio/mean": 1.0000667572021484, "sampling/importance_sampling_ratio/max": 1.0559391975402832, "entropy": 0.0018285177211510018, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8307228684425354, "reward_meter_mean": 0.9968674182891846, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8307228684425354, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1098.0} {"timestamp_utc": "2026-04-11T21:43:55Z", "mode": "train", "global_step": 1099, "epoch": 0.042438986716095146, "loss": -0.0331, "grad_norm": 6.627462863922119, "learning_rate": 6.672727272727273e-06, "num_tokens": 2366553.0, "completions/mean_length": 48.375, "completions/min_length": 46.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.375, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.18795685470104218, "rewards/meter/std": 0.3101930618286133, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.18795685470104218, "rewards/total_composite/std": 0.3101930618286133, "reward": 0.18795685470104218, "reward_std": 0.3101930618286133, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010162265971302986, "sampling/sampling_logp_difference/max": 0.9485464096069336, "sampling/importance_sampling_ratio/min": 0.3873036205768585, "sampling/importance_sampling_ratio/mean": 1.0062874555587769, "sampling/importance_sampling_ratio/max": 1.6379154920578003, "entropy": 0.030019621714018285, "clip_ratio/low_mean": 0.00657894741743803, "clip_ratio/low_min": 0.00657894741743803, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.00657894741743803, "reward_total_mean": 0.18795685470104218, "reward_meter_mean": 0.18795685470104218, "reward_meter_std": 0.3101930618286133, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.18795685470104218, "reward_total_composite_std": 0.3101930618286133, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1099.0} {"timestamp_utc": "2026-04-11T21:44:01Z", "mode": "train", "global_step": 1100, "epoch": 0.04247760271856657, "loss": -0.0306, "grad_norm": 2.0620083808898926, "learning_rate": 6.66969696969697e-06, "num_tokens": 2369329.0, "completions/mean_length": 156.0, "completions/min_length": 145.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 156.0, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.9976032972335815, "rewards/meter/std": 0.0007398881716653705, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7980826497077942, "rewards/total_composite/std": 0.0005919100949540734, "reward": 0.7980826497077942, "reward_std": 0.0005919178947806358, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003971020225435495, "sampling/sampling_logp_difference/max": 0.6630861759185791, "sampling/importance_sampling_ratio/min": 0.5152587294578552, "sampling/importance_sampling_ratio/mean": 1.0009613037109375, "sampling/importance_sampling_ratio/max": 1.730522871017456, "entropy": 0.030738614965230227, "clip_ratio/low_mean": 0.0016737572732381523, "clip_ratio/low_min": 0.0016737572732381523, "clip_ratio/high_mean": 0.0015337422955781221, "clip_ratio/high_max": 0.0015337422955781221, "clip_ratio/region_mean": 0.0032074995688162744, "reward_total_mean": 0.7980826497077942, "reward_meter_mean": 0.9976032972335815, "reward_meter_std": 0.0007398881716653705, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7980826497077942, "reward_total_composite_std": 0.0005919100949540734, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1100.0} {"timestamp_utc": "2026-04-11T21:45:15Z", "mode": "eval", "global_step": 1100, "epoch": 0.04247760271856657, "eval_loss": NaN, "eval_runtime": 73.9179, "eval_samples_per_second": 1.407, "eval_steps_per_second": 0.176, "eval_num_tokens": 2369329.0, "eval_completions/mean_length": 177.95192307692307, "eval_completions/min_length": 57.38461538461539, "eval_completions/max_length": 390.15384615384613, "eval_completions/clipped_ratio": 0.028846153846153848, "eval_completions/mean_terminated_length": 167.59066126896784, "eval_completions/min_terminated_length": 57.38461538461539, "eval_completions/max_terminated_length": 341.7692307692308, "eval_rewards/meter/mean": 0.6938754916191101, "eval_rewards/meter/std": 0.4057489140675618, "eval_rewards/count_adherence/mean": 0.8606285040195172, "eval_rewards/count_adherence/std": 0.14204404417138833, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/total_composite/mean": 0.5808727557842548, "eval_rewards/total_composite/std": 0.372596684556741, "eval_reward": 0.5808727557842548, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0031142185639160182, "eval_sampling/sampling_logp_difference/max": 0.6837713030668405, "eval_sampling/importance_sampling_ratio/min": 0.542173194197508, "eval_sampling/importance_sampling_ratio/mean": 1.000536437218006, "eval_sampling/importance_sampling_ratio/max": 1.281661501297584, "eval_entropy": 0.022705065874526136, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5808727557842548, "eval_reward_meter_mean": 0.6938754916191101, "eval_reward_meter_std": 0.4057489140675618, "eval_reward_count_adherence_mean": 0.8606285040195172, "eval_reward_count_adherence_std": 0.14204404417138833, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_total_composite_mean": 0.5808727557842548, "eval_reward_total_composite_std": 0.372596684556741, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1100.0} {"timestamp_utc": "2026-04-11T21:45:24Z", "mode": "train", "global_step": 1101, "epoch": 0.042516218721038, "loss": -0.2286, "grad_norm": 5.9600019454956055, "learning_rate": 6.666666666666667e-06, "num_tokens": 2371113.0, "completions/mean_length": 62.0, "completions/min_length": 18.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.772131621837616, "rewards/meter/std": 0.34598708152770996, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.772131621837616, "rewards/total_composite/std": 0.34598708152770996, "reward": 0.772131621837616, "reward_std": 0.34598708152770996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040163177996873856, "sampling/sampling_logp_difference/max": 1.5425870418548584, "sampling/importance_sampling_ratio/min": 0.2138272076845169, "sampling/importance_sampling_ratio/mean": 1.0026087760925293, "sampling/importance_sampling_ratio/max": 1.7506089210510254, "entropy": 0.2632972849532962, "clip_ratio/low_mean": 0.01245915051549673, "clip_ratio/low_min": 0.01245915051549673, "clip_ratio/high_mean": 0.023881223052740097, "clip_ratio/high_max": 0.023881223052740097, "clip_ratio/region_mean": 0.03634037356823683, "reward_total_mean": 0.772131621837616, "reward_meter_mean": 0.772131621837616, "reward_meter_std": 0.34598708152770996, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.772131621837616, "reward_total_composite_std": 0.34598708152770996, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1101.0} {"timestamp_utc": "2026-04-11T21:45:29Z", "mode": "train", "global_step": 1102, "epoch": 0.042554834723509426, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.663636363636365e-06, "num_tokens": 2372961.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00010852154809981585, "sampling/sampling_logp_difference/max": 0.007473420351743698, "sampling/importance_sampling_ratio/min": 0.9969366788864136, "sampling/importance_sampling_ratio/mean": 1.0000767707824707, "sampling/importance_sampling_ratio/max": 1.007501482963562, "entropy": 0.0009841258579399437, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1102.0} {"timestamp_utc": "2026-04-11T21:45:35Z", "mode": "train", "global_step": 1103, "epoch": 0.04259345072598085, "loss": -0.014, "grad_norm": 1.3136855363845825, "learning_rate": 6.660606060606061e-06, "num_tokens": 2375603.0, "completions/mean_length": 139.25, "completions/min_length": 137.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.25, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9981851577758789, "rewards/meter/std": 0.00017230756930075586, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9981851577758789, "rewards/total_composite/std": 0.00017230756930075586, "reward": 0.9981851577758789, "reward_std": 0.00017229857621714473, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0030210025142878294, "sampling/sampling_logp_difference/max": 0.3306952714920044, "sampling/importance_sampling_ratio/min": 0.7184240818023682, "sampling/importance_sampling_ratio/mean": 0.9999781250953674, "sampling/importance_sampling_ratio/max": 1.2929089069366455, "entropy": 0.021428768755868077, "clip_ratio/low_mean": 0.0036231884732842445, "clip_ratio/low_min": 0.0036231884732842445, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0036231884732842445, "reward_total_mean": 0.9981851577758789, "reward_meter_mean": 0.9981851577758789, "reward_meter_std": 0.00017230756930075586, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9981851577758789, "reward_total_composite_std": 0.00017230756930075586, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1103.0} {"timestamp_utc": "2026-04-11T21:45:42Z", "mode": "train", "global_step": 1104, "epoch": 0.042632066728452274, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.657575757575758e-06, "num_tokens": 2379419.0, "completions/mean_length": 271.0, "completions/min_length": 271.0, "completions/max_length": 271.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 271.0, "completions/min_terminated_length": 271.0, "completions/max_terminated_length": 271.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.8181818127632141, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8170298933982849, "rewards/total_composite/std": 0.0, "reward": 0.8170298933982849, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00010036063758889213, "sampling/sampling_logp_difference/max": 0.026611091569066048, "sampling/importance_sampling_ratio/min": 0.9737398624420166, "sampling/importance_sampling_ratio/mean": 1.0000735521316528, "sampling/importance_sampling_ratio/max": 1.0152407884597778, "entropy": 0.0003895306908816565, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8170298933982849, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.8181818127632141, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8170298933982849, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1104.0} {"timestamp_utc": "2026-04-11T21:45:47Z", "mode": "train", "global_step": 1105, "epoch": 0.0426706827309237, "loss": -0.063, "grad_norm": 9.432920455932617, "learning_rate": 6.654545454545455e-06, "num_tokens": 2380899.0, "completions/mean_length": 29.0, "completions/min_length": 26.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9883742928504944, "rewards/meter/std": 0.013260845094919205, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9883742928504944, "rewards/total_composite/std": 0.013260845094919205, "reward": 0.9883742928504944, "reward_std": 0.013260837644338608, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031237933784723282, "sampling/sampling_logp_difference/max": 1.4528617858886719, "sampling/importance_sampling_ratio/min": 0.25039586424827576, "sampling/importance_sampling_ratio/mean": 1.0062172412872314, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14257134380750358, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.008342602755874395, "clip_ratio/high_max": 0.008342602755874395, "clip_ratio/region_mean": 0.008342602755874395, "reward_total_mean": 0.9883742928504944, "reward_meter_mean": 0.9883742928504944, "reward_meter_std": 0.013260845094919205, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9883742928504944, "reward_total_composite_std": 0.013260845094919205, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1105.0} {"timestamp_utc": "2026-04-11T21:45:52Z", "mode": "train", "global_step": 1106, "epoch": 0.04270929873339512, "loss": 0.015, "grad_norm": 3.427839756011963, "learning_rate": 6.651515151515152e-06, "num_tokens": 2383097.0, "completions/mean_length": 111.75, "completions/min_length": 105.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.75, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.4524507522583008, "rewards/meter/std": 0.2328236848115921, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4524507522583008, "rewards/total_composite/std": 0.2328236848115921, "reward": 0.4524507522583008, "reward_std": 0.2328236848115921, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020164301618933678, "sampling/sampling_logp_difference/max": 1.8077073097229004, "sampling/importance_sampling_ratio/min": 0.1640297770500183, "sampling/importance_sampling_ratio/mean": 0.99989914894104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07199247367680073, "clip_ratio/low_mean": 0.0055914610857144, "clip_ratio/low_min": 0.0055914610857144, "clip_ratio/high_mean": 0.011224456364288926, "clip_ratio/high_max": 0.011224456364288926, "clip_ratio/region_mean": 0.016815917450003326, "reward_total_mean": 0.4524507522583008, "reward_meter_mean": 0.4524507522583008, "reward_meter_std": 0.2328236848115921, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4524507522583008, "reward_total_composite_std": 0.2328236848115921, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1106.0} {"timestamp_utc": "2026-04-11T21:45:57Z", "mode": "train", "global_step": 1107, "epoch": 0.042747914735866546, "loss": 0.0209, "grad_norm": 6.449989318847656, "learning_rate": 6.6484848484848485e-06, "num_tokens": 2384709.0, "completions/mean_length": 50.5, "completions/min_length": 48.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.5, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9843699932098389, "rewards/meter/std": 0.004019024316221476, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9843699932098389, "rewards/total_composite/std": 0.004019024316221476, "reward": 0.9843699932098389, "reward_std": 0.00401902524754405, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03694266453385353, "sampling/sampling_logp_difference/max": 2.5510663986206055, "sampling/importance_sampling_ratio/min": 0.07799844443798065, "sampling/importance_sampling_ratio/mean": 0.9934430718421936, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13322050124406815, "clip_ratio/low_mean": 0.004950980423018336, "clip_ratio/low_min": 0.004950980423018336, "clip_ratio/high_mean": 0.012708333320915699, "clip_ratio/high_max": 0.012708333320915699, "clip_ratio/region_mean": 0.017659313743934035, "reward_total_mean": 0.9843699932098389, "reward_meter_mean": 0.9843699932098389, "reward_meter_std": 0.004019024316221476, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9843699932098389, "reward_total_composite_std": 0.004019024316221476, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1107.0} {"timestamp_utc": "2026-04-11T21:46:03Z", "mode": "train", "global_step": 1108, "epoch": 0.04278653073833797, "loss": 0.0137, "grad_norm": 11.623807907104492, "learning_rate": 6.645454545454546e-06, "num_tokens": 2386216.0, "completions/mean_length": 26.375, "completions/min_length": 24.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9814929962158203, "rewards/meter/std": 0.008537075482308865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9814929962158203, "rewards/total_composite/std": 0.008537075482308865, "reward": 0.9814929962158203, "reward_std": 0.00853707455098629, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03857656940817833, "sampling/sampling_logp_difference/max": 1.7021937370300293, "sampling/importance_sampling_ratio/min": 0.1822832077741623, "sampling/importance_sampling_ratio/mean": 0.9879005551338196, "sampling/importance_sampling_ratio/max": 1.245644450187683, "entropy": 0.19634724967181683, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.03787647606804967, "clip_ratio/high_max": 0.03787647606804967, "clip_ratio/region_mean": 0.03787647606804967, "reward_total_mean": 0.9814929962158203, "reward_meter_mean": 0.9814929962158203, "reward_meter_std": 0.008537075482308865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9814929962158203, "reward_total_composite_std": 0.008537075482308865, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1108.0} {"timestamp_utc": "2026-04-11T21:46:07Z", "mode": "train", "global_step": 1109, "epoch": 0.042825146740809394, "loss": -0.0099, "grad_norm": 2.722727060317993, "learning_rate": 6.642424242424242e-06, "num_tokens": 2388091.0, "completions/mean_length": 72.375, "completions/min_length": 70.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9982842206954956, "rewards/meter/std": 0.00037441932363435626, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9982842206954956, "rewards/total_composite/std": 0.00037441932363435626, "reward": 0.9982842206954956, "reward_std": 0.00037443003384396434, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004766764119267464, "sampling/sampling_logp_difference/max": 0.5304169654846191, "sampling/importance_sampling_ratio/min": 0.5883596539497375, "sampling/importance_sampling_ratio/mean": 1.000014066696167, "sampling/importance_sampling_ratio/max": 1.271781086921692, "entropy": 0.029869895428419113, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/high_mean": 0.0017123287543654442, "clip_ratio/high_max": 0.0017123287543654442, "clip_ratio/region_mean": 0.005283757345750928, "reward_total_mean": 0.9982842206954956, "reward_meter_mean": 0.9982842206954956, "reward_meter_std": 0.00037441932363435626, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9982842206954956, "reward_total_composite_std": 0.00037441932363435626, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1109.0} {"timestamp_utc": "2026-04-11T21:46:12Z", "mode": "train", "global_step": 1110, "epoch": 0.04286376274328082, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.63939393939394e-06, "num_tokens": 2389963.0, "completions/mean_length": 62.0, "completions/min_length": 62.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9959487915039062, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9959487915039062, "rewards/total_composite/std": 0.0, "reward": 0.9959487915039062, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00021878920961171389, "sampling/sampling_logp_difference/max": 0.0036456617526710033, "sampling/importance_sampling_ratio/min": 0.9998347759246826, "sampling/importance_sampling_ratio/mean": 1.000218152999878, "sampling/importance_sampling_ratio/max": 1.0036523342132568, "entropy": 0.0016348987410310656, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9959487915039062, "reward_meter_mean": 0.9959487915039062, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9959487915039062, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1110.0} {"timestamp_utc": "2026-04-11T21:46:17Z", "mode": "train", "global_step": 1111, "epoch": 0.04290237874575224, "loss": -0.0114, "grad_norm": 1.1772817373275757, "learning_rate": 6.6363636363636375e-06, "num_tokens": 2391880.0, "completions/mean_length": 81.625, "completions/min_length": 78.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.625, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.8804362416267395, "rewards/meter/std": 0.020153922960162163, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8804362416267395, "rewards/total_composite/std": 0.020153922960162163, "reward": 0.8804362416267395, "reward_std": 0.02015392854809761, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012632966041564941, "sampling/sampling_logp_difference/max": 2.2472591400146484, "sampling/importance_sampling_ratio/min": 0.10568850487470627, "sampling/importance_sampling_ratio/mean": 1.0001097917556763, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03438670525792986, "clip_ratio/low_mean": 0.0016025641234591603, "clip_ratio/low_min": 0.0016025641234591603, "clip_ratio/high_mean": 0.009184887632727623, "clip_ratio/high_max": 0.009184887632727623, "clip_ratio/region_mean": 0.010787451756186783, "reward_total_mean": 0.8804362416267395, "reward_meter_mean": 0.8804362416267395, "reward_meter_std": 0.020153922960162163, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8804362416267395, "reward_total_composite_std": 0.020153922960162163, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1111.0} {"timestamp_utc": "2026-04-11T21:46:22Z", "mode": "train", "global_step": 1112, "epoch": 0.04294099474822367, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.633333333333334e-06, "num_tokens": 2393504.0, "completions/mean_length": 57.0, "completions/min_length": 57.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8246031403541565, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8246031403541565, "rewards/total_composite/std": 0.0, "reward": 0.8246031403541565, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0025924842339009047, "sampling/sampling_logp_difference/max": 0.2582397758960724, "sampling/importance_sampling_ratio/min": 0.7724100351333618, "sampling/importance_sampling_ratio/mean": 0.9993739724159241, "sampling/importance_sampling_ratio/max": 1.0344194173812866, "entropy": 0.010429158399347216, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8246031403541565, "reward_meter_mean": 0.8246031403541565, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8246031403541565, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1112.0} {"timestamp_utc": "2026-04-11T21:46:27Z", "mode": "train", "global_step": 1113, "epoch": 0.04297961075069509, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.630303030303031e-06, "num_tokens": 2395152.0, "completions/mean_length": 57.0, "completions/min_length": 57.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8420836925506592, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8420836925506592, "rewards/total_composite/std": 0.0, "reward": 0.8420836925506592, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006502047181129456, "sampling/sampling_logp_difference/max": 0.022446556016802788, "sampling/importance_sampling_ratio/min": 0.9778034687042236, "sampling/importance_sampling_ratio/mean": 1.0003682374954224, "sampling/importance_sampling_ratio/max": 1.0151196718215942, "entropy": 0.004856940591707826, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8420836925506592, "reward_meter_mean": 0.8420836925506592, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8420836925506592, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1113.0} {"timestamp_utc": "2026-04-11T21:46:33Z", "mode": "train", "global_step": 1114, "epoch": 0.043018226753166515, "loss": -0.0013, "grad_norm": 1.41348397731781, "learning_rate": 6.627272727272728e-06, "num_tokens": 2397325.0, "completions/mean_length": 90.625, "completions/min_length": 89.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.625, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.9967718124389648, "rewards/meter/std": 0.00015542571782134473, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9967718124389648, "rewards/total_composite/std": 0.00015542571782134473, "reward": 0.9967718124389648, "reward_std": 0.00015544342750217766, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009261665865778923, "sampling/sampling_logp_difference/max": 1.2413575649261475, "sampling/importance_sampling_ratio/min": 0.28899163007736206, "sampling/importance_sampling_ratio/mean": 0.9978534579277039, "sampling/importance_sampling_ratio/max": 1.1216578483581543, "entropy": 0.030776873929426074, "clip_ratio/low_mean": 0.0027173913549631834, "clip_ratio/low_min": 0.0027173913549631834, "clip_ratio/high_mean": 0.0013888889225199819, "clip_ratio/high_max": 0.0013888889225199819, "clip_ratio/region_mean": 0.004106280277483165, "reward_total_mean": 0.9967718124389648, "reward_meter_mean": 0.9967718124389648, "reward_meter_std": 0.00015542571782134473, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9967718124389648, "reward_total_composite_std": 0.00015542571782134473, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1114.0} {"timestamp_utc": "2026-04-11T21:46:39Z", "mode": "train", "global_step": 1115, "epoch": 0.04305684275563794, "loss": 0.0356, "grad_norm": 1.2258256673812866, "learning_rate": 6.624242424242425e-06, "num_tokens": 2399993.0, "completions/mean_length": 136.5, "completions/min_length": 132.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.5, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.8965202569961548, "rewards/meter/std": 0.10726836323738098, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7471002340316772, "rewards/total_composite/std": 0.08939030766487122, "reward": 0.7471002340316772, "reward_std": 0.08939030766487122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0015983968041837215, "sampling/sampling_logp_difference/max": 0.3348802328109741, "sampling/importance_sampling_ratio/min": 0.7790073156356812, "sampling/importance_sampling_ratio/mean": 1.0006446838378906, "sampling/importance_sampling_ratio/max": 1.3977729082107544, "entropy": 0.009146204334683716, "clip_ratio/low_mean": 0.0016666667070239782, "clip_ratio/low_min": 0.0016666667070239782, "clip_ratio/high_mean": 0.0009469697251915932, "clip_ratio/high_max": 0.0009469697251915932, "clip_ratio/region_mean": 0.0026136364322155714, "reward_total_mean": 0.7471002340316772, "reward_meter_mean": 0.8965202569961548, "reward_meter_std": 0.10726836323738098, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7471002340316772, "reward_total_composite_std": 0.08939030766487122, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1115.0} {"timestamp_utc": "2026-04-11T21:46:45Z", "mode": "train", "global_step": 1116, "epoch": 0.04309545875810936, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.621212121212121e-06, "num_tokens": 2402617.0, "completions/mean_length": 151.0, "completions/min_length": 151.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.0, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8321600556373596, "rewards/total_composite/std": 0.0, "reward": 0.8321600556373596, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 4.834596620639786e-05, "sampling/sampling_logp_difference/max": 0.005399270448833704, "sampling/importance_sampling_ratio/min": 0.9998179078102112, "sampling/importance_sampling_ratio/mean": 1.0000474452972412, "sampling/importance_sampling_ratio/max": 1.0054138898849487, "entropy": 0.0003524243838910479, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8321600556373596, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8321600556373596, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1116.0} {"timestamp_utc": "2026-04-11T21:46:49Z", "mode": "train", "global_step": 1117, "epoch": 0.04313407476058079, "loss": -0.0125, "grad_norm": 2.3175323009490967, "learning_rate": 6.618181818181819e-06, "num_tokens": 2404357.0, "completions/mean_length": 61.5, "completions/min_length": 61.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9971427321434021, "rewards/meter/std": 0.0005570190260186791, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9971427321434021, "rewards/total_composite/std": 0.0005570190260186791, "reward": 0.9971427321434021, "reward_std": 0.0005570190260186791, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0036074817180633545, "sampling/sampling_logp_difference/max": 0.32697343826293945, "sampling/importance_sampling_ratio/min": 0.7471194863319397, "sampling/importance_sampling_ratio/mean": 1.0014525651931763, "sampling/importance_sampling_ratio/max": 1.386764645576477, "entropy": 0.022033903398551047, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.003968254197388887, "clip_ratio/high_max": 0.003968254197388887, "clip_ratio/region_mean": 0.003968254197388887, "reward_total_mean": 0.9971427321434021, "reward_meter_mean": 0.9971427321434021, "reward_meter_std": 0.0005570190260186791, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9971427321434021, "reward_total_composite_std": 0.0005570190260186791, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1117.0} {"timestamp_utc": "2026-04-11T21:46:54Z", "mode": "train", "global_step": 1118, "epoch": 0.04317269076305221, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.615151515151516e-06, "num_tokens": 2405789.0, "completions/mean_length": 30.0, "completions/min_length": 30.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9949287176132202, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9949287176132202, "rewards/total_composite/std": 0.0, "reward": 0.9949287176132202, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.004324022214859724, "sampling/sampling_logp_difference/max": 0.18327409029006958, "sampling/importance_sampling_ratio/min": 0.8582614660263062, "sampling/importance_sampling_ratio/mean": 1.0010007619857788, "sampling/importance_sampling_ratio/max": 1.2011436223983765, "entropy": 0.025725294835865498, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9949287176132202, "reward_meter_mean": 0.9949287176132202, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9949287176132202, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1118.0} {"timestamp_utc": "2026-04-11T21:46:59Z", "mode": "train", "global_step": 1119, "epoch": 0.043211306765523635, "loss": 0.0481, "grad_norm": 2.774005889892578, "learning_rate": 6.612121212121213e-06, "num_tokens": 2408465.0, "completions/mean_length": 133.5, "completions/min_length": 126.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.5, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.564446747303009, "rewards/meter/std": 0.4670409858226776, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5460528135299683, "rewards/total_composite/std": 0.4622097909450531, "reward": 0.5460528135299683, "reward_std": 0.4622097909450531, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013703513890504837, "sampling/sampling_logp_difference/max": 2.1475610733032227, "sampling/importance_sampling_ratio/min": 0.22612382471561432, "sampling/importance_sampling_ratio/mean": 1.0015851259231567, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.045366469537839293, "clip_ratio/low_mean": 0.0036062682047486305, "clip_ratio/low_min": 0.0036062682047486305, "clip_ratio/high_mean": 0.0029761906480416656, "clip_ratio/high_max": 0.0029761906480416656, "clip_ratio/region_mean": 0.006582458852790296, "reward_total_mean": 0.5460528135299683, "reward_meter_mean": 0.564446747303009, "reward_meter_std": 0.4670409858226776, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5460528135299683, "reward_total_composite_std": 0.4622097909450531, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1119.0} {"timestamp_utc": "2026-04-11T21:47:05Z", "mode": "train", "global_step": 1120, "epoch": 0.04324992276799506, "loss": -0.0001, "grad_norm": 0.031646136194467545, "learning_rate": 6.609090909090909e-06, "num_tokens": 2411344.0, "completions/mean_length": 162.875, "completions/min_length": 162.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 162.875, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.998467206954956, "rewards/meter/std": 2.982700607390143e-06, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6656447649002075, "rewards/total_composite/std": 1.9757069367187796e-06, "reward": 0.6656447649002075, "reward_std": 1.9751287254621275e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0016431320691481233, "sampling/sampling_logp_difference/max": 0.9145364761352539, "sampling/importance_sampling_ratio/min": 0.4007023572921753, "sampling/importance_sampling_ratio/mean": 0.9996026754379272, "sampling/importance_sampling_ratio/max": 1.057082176208496, "entropy": 0.0052466230117715895, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.6656447649002075, "reward_meter_mean": 0.998467206954956, "reward_meter_std": 2.982700607390143e-06, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6656447649002075, "reward_total_composite_std": 1.9757069367187796e-06, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1120.0} {"timestamp_utc": "2026-04-11T21:47:11Z", "mode": "train", "global_step": 1121, "epoch": 0.043288538770466484, "loss": -0.0448, "grad_norm": 5.260746955871582, "learning_rate": 6.606060606060607e-06, "num_tokens": 2413261.0, "completions/mean_length": 72.625, "completions/min_length": 61.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.625, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.0058386242017149925, "rewards/meter/std": 0.002312840661033988, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0058386242017149925, "rewards/total_composite/std": 0.002312840661033988, "reward": 0.0058386242017149925, "reward_std": 0.0023128404282033443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02798517607152462, "sampling/sampling_logp_difference/max": 4.47673225402832, "sampling/importance_sampling_ratio/min": 0.011370508931577206, "sampling/importance_sampling_ratio/mean": 1.0017359256744385, "sampling/importance_sampling_ratio/max": 1.6017389297485352, "entropy": 0.10170990042388439, "clip_ratio/low_mean": 0.0016447368543595076, "clip_ratio/low_min": 0.0016447368543595076, "clip_ratio/high_mean": 0.006582458503544331, "clip_ratio/high_max": 0.006582458503544331, "clip_ratio/region_mean": 0.008227195357903838, "reward_total_mean": 0.0058386242017149925, "reward_meter_mean": 0.0058386242017149925, "reward_meter_std": 0.002312840661033988, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0058386242017149925, "reward_total_composite_std": 0.002312840661033988, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1121.0} {"timestamp_utc": "2026-04-11T21:47:18Z", "mode": "train", "global_step": 1122, "epoch": 0.04332715477293791, "loss": -0.0229, "grad_norm": 1.8486534357070923, "learning_rate": 6.603030303030303e-06, "num_tokens": 2416946.0, "completions/mean_length": 284.625, "completions/min_length": 256.0, "completions/max_length": 289.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 284.625, "completions/min_terminated_length": 256.0, "completions/max_terminated_length": 289.0, "rewards/meter/mean": 0.0074860588647425175, "rewards/meter/std": 0.003230373840779066, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0059888469986617565, "rewards/total_composite/std": 0.0025842993054538965, "reward": 0.0059888469986617565, "reward_std": 0.002584299072623253, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004391747061163187, "sampling/sampling_logp_difference/max": 1.2448701858520508, "sampling/importance_sampling_ratio/min": 0.28797829151153564, "sampling/importance_sampling_ratio/mean": 1.0002622604370117, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.018305832403711975, "clip_ratio/low_mean": 0.0008650519303046167, "clip_ratio/low_min": 0.0008650519303046167, "clip_ratio/high_mean": 0.002601163083454594, "clip_ratio/high_max": 0.002601163083454594, "clip_ratio/region_mean": 0.0034662150137592107, "reward_total_mean": 0.0059888469986617565, "reward_meter_mean": 0.0074860588647425175, "reward_meter_std": 0.003230373840779066, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0059888469986617565, "reward_total_composite_std": 0.0025842993054538965, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1122.0} {"timestamp_utc": "2026-04-11T21:47:23Z", "mode": "train", "global_step": 1123, "epoch": 0.04336577077540933, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.600000000000001e-06, "num_tokens": 2418666.0, "completions/mean_length": 49.0, "completions/min_length": 49.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9879102110862732, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9879102110862732, "rewards/total_composite/std": 0.0, "reward": 0.9879102110862732, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.002916015451774001, "sampling/sampling_logp_difference/max": 0.1805974692106247, "sampling/importance_sampling_ratio/min": 0.8347713351249695, "sampling/importance_sampling_ratio/mean": 1.001497507095337, "sampling/importance_sampling_ratio/max": 1.0804226398468018, "entropy": 0.024839470628648996, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9879102110862732, "reward_meter_mean": 0.9879102110862732, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9879102110862732, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1123.0} {"timestamp_utc": "2026-04-11T21:47:28Z", "mode": "train", "global_step": 1124, "epoch": 0.043404386777880756, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.596969696969698e-06, "num_tokens": 2420498.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00012347516894806176, "sampling/sampling_logp_difference/max": 0.005228639580309391, "sampling/importance_sampling_ratio/min": 0.9982930421829224, "sampling/importance_sampling_ratio/mean": 1.0001134872436523, "sampling/importance_sampling_ratio/max": 1.0052423477172852, "entropy": 0.002108675726049114, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1124.0} {"timestamp_utc": "2026-04-11T21:47:36Z", "mode": "train", "global_step": 1125, "epoch": 0.04344300278035218, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.593939393939395e-06, "num_tokens": 2424688.0, "completions/mean_length": 295.75, "completions/min_length": 289.0, "completions/max_length": 307.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 295.75, "completions/min_terminated_length": 289.0, "completions/max_terminated_length": 307.0, "rewards/meter/mean": 0.9984642863273621, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.7272727489471436, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7261558175086975, "rewards/total_composite/std": 0.0, "reward": 0.7261558175086975, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0026164765004068613, "sampling/sampling_logp_difference/max": 1.7056536674499512, "sampling/importance_sampling_ratio/min": 0.18165360391139984, "sampling/importance_sampling_ratio/mean": 0.9990125894546509, "sampling/importance_sampling_ratio/max": 1.2641671895980835, "entropy": 0.0053885679808445275, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7261558175086975, "reward_meter_mean": 0.9984642863273621, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.7272727489471436, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7261558175086975, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1125.0} {"timestamp_utc": "2026-04-11T21:47:40Z", "mode": "train", "global_step": 1126, "epoch": 0.043481618782823604, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.590909090909091e-06, "num_tokens": 2426392.0, "completions/mean_length": 60.0, "completions/min_length": 60.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9865735769271851, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9865735769271851, "rewards/total_composite/std": 0.0, "reward": 0.9865735769271851, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.011119093745946884, "sampling/sampling_logp_difference/max": 3.5465991497039795, "sampling/importance_sampling_ratio/min": 0.02882249280810356, "sampling/importance_sampling_ratio/mean": 0.9979950189590454, "sampling/importance_sampling_ratio/max": 1.2449251413345337, "entropy": 0.01859230361878872, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9865735769271851, "reward_meter_mean": 0.9865735769271851, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9865735769271851, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1126.0} {"timestamp_utc": "2026-04-11T21:47:45Z", "mode": "train", "global_step": 1127, "epoch": 0.04352023478529503, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.5878787878787885e-06, "num_tokens": 2428384.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 9.980119648389518e-05, "sampling/sampling_logp_difference/max": 0.006944278255105019, "sampling/importance_sampling_ratio/min": 0.9992338418960571, "sampling/importance_sampling_ratio/mean": 1.0000923871994019, "sampling/importance_sampling_ratio/max": 1.0069684982299805, "entropy": 0.0010559483562246896, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1127.0} {"timestamp_utc": "2026-04-11T21:47:55Z", "mode": "train", "global_step": 1128, "epoch": 0.04355885078776645, "loss": -0.0845, "grad_norm": 2.203399181365967, "learning_rate": 6.584848484848485e-06, "num_tokens": 2430432.0, "completions/mean_length": 147.0, "completions/min_length": 84.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 94.85714721679688, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.02705313451588154, "rewards/meter/std": 0.02497793175280094, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.018090665340423584, "rewards/total_composite/std": 0.01749170385301113, "reward": 0.018090665340423584, "reward_std": 0.01749170385301113, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014606590382754803, "sampling/sampling_logp_difference/max": 1.492810845375061, "sampling/importance_sampling_ratio/min": 0.3910922706127167, "sampling/importance_sampling_ratio/mean": 1.0034804344177246, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06913374830037355, "clip_ratio/low_mean": 0.0026178729021921754, "clip_ratio/low_min": 0.0026178729021921754, "clip_ratio/high_mean": 0.010015423293225467, "clip_ratio/high_max": 0.010015423293225467, "clip_ratio/region_mean": 0.012633296195417643, "reward_total_mean": 0.018090665340423584, "reward_meter_mean": 0.02705313451588154, "reward_meter_std": 0.02497793175280094, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.018090665340423584, "reward_total_composite_std": 0.01749170385301113, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1128.0} {"timestamp_utc": "2026-04-11T21:48:01Z", "mode": "train", "global_step": 1129, "epoch": 0.043597466790237877, "loss": -0.0474, "grad_norm": 4.438967227935791, "learning_rate": 6.581818181818182e-06, "num_tokens": 2432847.0, "completions/mean_length": 132.875, "completions/min_length": 124.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.875, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.010067595168948174, "rewards/meter/std": 0.012099563144147396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.010067595168948174, "rewards/total_composite/std": 0.012099563144147396, "reward": 0.010067595168948174, "reward_std": 0.012099562212824821, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03801698610186577, "sampling/sampling_logp_difference/max": 8.372949600219727, "sampling/importance_sampling_ratio/min": 0.00023103307466953993, "sampling/importance_sampling_ratio/mean": 1.0018302202224731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07547829509712756, "clip_ratio/low_mean": 0.002016128972172737, "clip_ratio/low_min": 0.002016128972172737, "clip_ratio/high_mean": 0.012722764164209366, "clip_ratio/high_max": 0.012722764164209366, "clip_ratio/region_mean": 0.014738893136382103, "reward_total_mean": 0.010067595168948174, "reward_meter_mean": 0.010067595168948174, "reward_meter_std": 0.012099563144147396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.010067595168948174, "reward_total_composite_std": 0.012099563144147396, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1129.0} {"timestamp_utc": "2026-04-11T21:48:07Z", "mode": "train", "global_step": 1130, "epoch": 0.0436360827927093, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.578787878787879e-06, "num_tokens": 2435391.0, "completions/mean_length": 137.0, "completions/min_length": 137.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.0, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.980923593044281, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7847388982772827, "rewards/total_composite/std": 0.0, "reward": 0.7847388982772827, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006849313504062593, "sampling/sampling_logp_difference/max": 0.4150339663028717, "sampling/importance_sampling_ratio/min": 0.6603178381919861, "sampling/importance_sampling_ratio/mean": 0.999580442905426, "sampling/importance_sampling_ratio/max": 1.011684775352478, "entropy": 0.005225834396696882, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7847388982772827, "reward_meter_mean": 0.980923593044281, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7847388982772827, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1130.0} {"timestamp_utc": "2026-04-11T21:48:17Z", "mode": "train", "global_step": 1131, "epoch": 0.043674698795180725, "loss": -0.1575, "grad_norm": 3.7873270511627197, "learning_rate": 6.575757575757577e-06, "num_tokens": 2437832.0, "completions/mean_length": 194.125, "completions/min_length": 132.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 148.71429443359375, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.03643738478422165, "rewards/meter/std": 0.02530757524073124, "rewards/count_adherence/mean": 0.7250000238418579, "rewards/count_adherence/std": 0.2121320515871048, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.0285042654722929, "rewards/total_composite/std": 0.02118096500635147, "reward": 0.0285042654722929, "reward_std": 0.02118096500635147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03201490640640259, "sampling/sampling_logp_difference/max": 2.680152654647827, "sampling/importance_sampling_ratio/min": 0.06855268776416779, "sampling/importance_sampling_ratio/mean": 0.9909701943397522, "sampling/importance_sampling_ratio/max": 1.878620982170105, "entropy": 0.055357005912810564, "clip_ratio/low_mean": 0.007927836035378277, "clip_ratio/low_min": 0.007927836035378277, "clip_ratio/high_mean": 0.00701042782748118, "clip_ratio/high_max": 0.00701042782748118, "clip_ratio/region_mean": 0.014938263862859458, "reward_total_mean": 0.0285042654722929, "reward_meter_mean": 0.03643738478422165, "reward_meter_std": 0.02530757524073124, "reward_count_adherence_mean": 0.7250000238418579, "reward_count_adherence_std": 0.2121320515871048, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.0285042654722929, "reward_total_composite_std": 0.02118096500635147, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1131.0} {"timestamp_utc": "2026-04-11T21:48:24Z", "mode": "train", "global_step": 1132, "epoch": 0.04371331479765215, "loss": -0.0223, "grad_norm": 2.2841248512268066, "learning_rate": 6.572727272727273e-06, "num_tokens": 2442085.0, "completions/mean_length": 292.625, "completions/min_length": 287.0, "completions/max_length": 302.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 292.625, "completions/min_terminated_length": 287.0, "completions/max_terminated_length": 302.0, "rewards/meter/mean": 0.9797583222389221, "rewards/meter/std": 4.679692574427463e-05, "rewards/count_adherence/mean": 0.7211538553237915, "rewards/count_adherence/std": 0.039811473339796066, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7065548896789551, "rewards/total_composite/std": 0.038970947265625, "reward": 0.7065548896789551, "reward_std": 0.038970947265625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0032520387321710587, "sampling/sampling_logp_difference/max": 1.7337863445281982, "sampling/importance_sampling_ratio/min": 0.1766144186258316, "sampling/importance_sampling_ratio/mean": 0.999126672744751, "sampling/importance_sampling_ratio/max": 1.5071643590927124, "entropy": 0.0024712488520890474, "clip_ratio/low_mean": 0.000871080148499459, "clip_ratio/low_min": 0.000871080148499459, "clip_ratio/high_mean": 0.0004139072843827307, "clip_ratio/high_max": 0.0004139072843827307, "clip_ratio/region_mean": 0.0012849874328821898, "reward_total_mean": 0.7065548896789551, "reward_meter_mean": 0.9797583222389221, "reward_meter_std": 4.679692574427463e-05, "reward_count_adherence_mean": 0.7211538553237915, "reward_count_adherence_std": 0.039811473339796066, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7065548896789551, "reward_total_composite_std": 0.038970947265625, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1132.0} {"timestamp_utc": "2026-04-11T21:48:29Z", "mode": "train", "global_step": 1133, "epoch": 0.04375193080012357, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.56969696969697e-06, "num_tokens": 2443613.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9951131939888, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9951131939888, "rewards/total_composite/std": 0.0, "reward": 0.9951131939888, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 6.63905157125555e-05, "sampling/sampling_logp_difference/max": 0.0017685489729046822, "sampling/importance_sampling_ratio/min": 1.0000001192092896, "sampling/importance_sampling_ratio/mean": 1.0000663995742798, "sampling/importance_sampling_ratio/max": 1.0017701387405396, "entropy": 0.0005665823264280334, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9951131939888, "reward_meter_mean": 0.9951131939888, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9951131939888, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1133.0} {"timestamp_utc": "2026-04-11T21:48:33Z", "mode": "train", "global_step": 1134, "epoch": 0.043790546802595, "loss": 0.0373, "grad_norm": 13.081120491027832, "learning_rate": 6.566666666666667e-06, "num_tokens": 2445010.0, "completions/mean_length": 30.625, "completions/min_length": 28.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.625, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.4127254784107208, "rewards/meter/std": 0.372755229473114, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4127254784107208, "rewards/total_composite/std": 0.372755229473114, "reward": 0.4127254784107208, "reward_std": 0.372755229473114, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.060395125299692154, "sampling/sampling_logp_difference/max": 1.3312345743179321, "sampling/importance_sampling_ratio/min": 0.2641509473323822, "sampling/importance_sampling_ratio/mean": 0.9982876181602478, "sampling/importance_sampling_ratio/max": 1.6932135820388794, "entropy": 0.2852039057761431, "clip_ratio/low_mean": 0.03217253182083368, "clip_ratio/low_min": 0.03217253182083368, "clip_ratio/high_mean": 0.012122844811528921, "clip_ratio/high_max": 0.012122844811528921, "clip_ratio/region_mean": 0.044295376632362604, "reward_total_mean": 0.4127254784107208, "reward_meter_mean": 0.4127254784107208, "reward_meter_std": 0.372755229473114, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4127254784107208, "reward_total_composite_std": 0.372755229473114, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1134.0} {"timestamp_utc": "2026-04-11T21:48:41Z", "mode": "train", "global_step": 1135, "epoch": 0.04382916280506642, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.563636363636364e-06, "num_tokens": 2449362.0, "completions/mean_length": 331.0, "completions/min_length": 331.0, "completions/max_length": 331.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 331.0, "completions/min_terminated_length": 331.0, "completions/max_terminated_length": 331.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.8461538553237915, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8449625372886658, "rewards/total_composite/std": 0.0, "reward": 0.8449625372886658, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 7.297201955225319e-05, "sampling/sampling_logp_difference/max": 0.017023995518684387, "sampling/importance_sampling_ratio/min": 0.9915810823440552, "sampling/importance_sampling_ratio/mean": 1.000062346458435, "sampling/importance_sampling_ratio/max": 1.017169713973999, "entropy": 0.0008706011249159928, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8449625372886658, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.8461538553237915, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8449625372886658, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1135.0} {"timestamp_utc": "2026-04-11T21:48:46Z", "mode": "train", "global_step": 1136, "epoch": 0.043867778807537845, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.56060606060606e-06, "num_tokens": 2451098.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00016016287554521114, "sampling/sampling_logp_difference/max": 0.008298270404338837, "sampling/importance_sampling_ratio/min": 0.9934648871421814, "sampling/importance_sampling_ratio/mean": 1.0001276731491089, "sampling/importance_sampling_ratio/max": 1.0083328485488892, "entropy": 0.0014408476781682111, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1136.0} {"timestamp_utc": "2026-04-11T21:48:51Z", "mode": "train", "global_step": 1137, "epoch": 0.04390639481000927, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.5575757575757585e-06, "num_tokens": 2452922.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00022661424009129405, "sampling/sampling_logp_difference/max": 0.011327030137181282, "sampling/importance_sampling_ratio/min": 0.994108259677887, "sampling/importance_sampling_ratio/mean": 1.0001767873764038, "sampling/importance_sampling_ratio/max": 1.0113914012908936, "entropy": 0.0018004292651312426, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1137.0} {"timestamp_utc": "2026-04-11T21:48:56Z", "mode": "train", "global_step": 1138, "epoch": 0.04394501081248069, "loss": 0.0154, "grad_norm": 4.488828182220459, "learning_rate": 6.554545454545455e-06, "num_tokens": 2454978.0, "completions/mean_length": 98.0, "completions/min_length": 97.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9884132146835327, "rewards/meter/std": 0.00012789461470674723, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9884132146835327, "rewards/total_composite/std": 0.00012789461470674723, "reward": 0.9884132146835327, "reward_std": 0.00012789761240128428, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00240718643181026, "sampling/sampling_logp_difference/max": 0.5080879926681519, "sampling/importance_sampling_ratio/min": 0.6016448736190796, "sampling/importance_sampling_ratio/mean": 0.9997903108596802, "sampling/importance_sampling_ratio/max": 1.0543768405914307, "entropy": 0.012036932515911758, "clip_ratio/low_mean": 0.0011904762359336019, "clip_ratio/low_min": 0.0011904762359336019, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0011904762359336019, "reward_total_mean": 0.9884132146835327, "reward_meter_mean": 0.9884132146835327, "reward_meter_std": 0.00012789461470674723, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9884132146835327, "reward_total_composite_std": 0.00012789461470674723, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1138.0} {"timestamp_utc": "2026-04-11T21:49:06Z", "mode": "train", "global_step": 1139, "epoch": 0.04398362681495212, "loss": -0.059, "grad_norm": 1.7216053009033203, "learning_rate": 6.551515151515152e-06, "num_tokens": 2456767.0, "completions/mean_length": 190.625, "completions/min_length": 76.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 83.5, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.15200825035572052, "rewards/meter/std": 0.18127578496932983, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.11745582520961761, "rewards/total_composite/std": 0.18092525005340576, "reward": 0.11745582520961761, "reward_std": 0.18092523515224457, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026299957185983658, "sampling/sampling_logp_difference/max": 1.0056216716766357, "sampling/importance_sampling_ratio/min": 0.5244839191436768, "sampling/importance_sampling_ratio/mean": 1.010207176208496, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13788712210953236, "clip_ratio/low_mean": 0.011396670481190085, "clip_ratio/low_min": 0.011396670481190085, "clip_ratio/high_mean": 0.0045437406515702605, "clip_ratio/high_max": 0.0045437406515702605, "clip_ratio/region_mean": 0.015940411132760346, "reward_total_mean": 0.11745582520961761, "reward_meter_mean": 0.15200825035572052, "reward_meter_std": 0.18127578496932983, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.11745582520961761, "reward_total_composite_std": 0.18092525005340576, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1139.0} {"timestamp_utc": "2026-04-11T21:49:11Z", "mode": "train", "global_step": 1140, "epoch": 0.04402224281742354, "loss": 0.0628, "grad_norm": 4.255577564239502, "learning_rate": 6.5484848484848494e-06, "num_tokens": 2458764.0, "completions/mean_length": 74.625, "completions/min_length": 72.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.625, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9828473925590515, "rewards/meter/std": 0.015827437862753868, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9828473925590515, "rewards/total_composite/std": 0.015827437862753868, "reward": 0.9828473925590515, "reward_std": 0.015827437862753868, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006431057117879391, "sampling/sampling_logp_difference/max": 0.9595508575439453, "sampling/importance_sampling_ratio/min": 0.3830648958683014, "sampling/importance_sampling_ratio/mean": 1.0000231266021729, "sampling/importance_sampling_ratio/max": 1.8631874322891235, "entropy": 0.020905857323668897, "clip_ratio/low_mean": 0.0028409091755747795, "clip_ratio/low_min": 0.0028409091755747795, "clip_ratio/high_mean": 0.0017123287543654442, "clip_ratio/high_max": 0.0017123287543654442, "clip_ratio/region_mean": 0.004553237929940224, "reward_total_mean": 0.9828473925590515, "reward_meter_mean": 0.9828473925590515, "reward_meter_std": 0.015827437862753868, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9828473925590515, "reward_total_composite_std": 0.015827437862753868, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1140.0} {"timestamp_utc": "2026-04-11T21:49:17Z", "mode": "train", "global_step": 1141, "epoch": 0.044060858819894966, "loss": -0.0002, "grad_norm": 0.48495835065841675, "learning_rate": 6.545454545454546e-06, "num_tokens": 2461268.0, "completions/mean_length": 145.0, "completions/min_length": 145.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.0, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9984445571899414, "rewards/meter/std": 4.436980452737771e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984445571899414, "rewards/total_composite/std": 4.436980452737771e-05, "reward": 0.9984445571899414, "reward_std": 4.4368985982146114e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0013455023290589452, "sampling/sampling_logp_difference/max": 0.30580687522888184, "sampling/importance_sampling_ratio/min": 0.7612590193748474, "sampling/importance_sampling_ratio/mean": 1.000628113746643, "sampling/importance_sampling_ratio/max": 1.357720136642456, "entropy": 0.007926426886115223, "clip_ratio/low_mean": 0.0008620689623057842, "clip_ratio/low_min": 0.0008620689623057842, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0008620689623057842, "reward_total_mean": 0.9984445571899414, "reward_meter_mean": 0.9984445571899414, "reward_meter_std": 4.436980452737771e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984445571899414, "reward_total_composite_std": 4.436980452737771e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1141.0} {"timestamp_utc": "2026-04-11T21:49:22Z", "mode": "train", "global_step": 1142, "epoch": 0.04409947482236639, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.542424242424243e-06, "num_tokens": 2463852.0, "completions/mean_length": 145.0, "completions/min_length": 145.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.0, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9984685182571411, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984685182571411, "rewards/total_composite/std": 0.0, "reward": 0.9984685182571411, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0007693552761338651, "sampling/sampling_logp_difference/max": 0.05545756220817566, "sampling/importance_sampling_ratio/min": 0.9495499134063721, "sampling/importance_sampling_ratio/mean": 1.0006191730499268, "sampling/importance_sampling_ratio/max": 1.0570241212844849, "entropy": 0.006758017058018595, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9984685182571411, "reward_meter_mean": 0.9984685182571411, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984685182571411, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1142.0} {"timestamp_utc": "2026-04-11T21:49:28Z", "mode": "train", "global_step": 1143, "epoch": 0.044138090824837814, "loss": 0.0079, "grad_norm": 7.574105739593506, "learning_rate": 6.5393939393939395e-06, "num_tokens": 2465932.0, "completions/mean_length": 98.0, "completions/min_length": 97.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9961534738540649, "rewards/meter/std": 0.0004939694190397859, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9961534738540649, "rewards/total_composite/std": 0.0004939694190397859, "reward": 0.9961534738540649, "reward_std": 0.0004939701757393777, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06125796586275101, "sampling/sampling_logp_difference/max": 11.969964027404785, "sampling/importance_sampling_ratio/min": 6.331559234240558e-06, "sampling/importance_sampling_ratio/mean": 0.9875028729438782, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03108868794515729, "clip_ratio/low_mean": 0.010165429906919599, "clip_ratio/low_min": 0.010165429906919599, "clip_ratio/high_mean": 0.01533242140430957, "clip_ratio/high_max": 0.01533242140430957, "clip_ratio/region_mean": 0.02549785131122917, "reward_total_mean": 0.9961534738540649, "reward_meter_mean": 0.9961534738540649, "reward_meter_std": 0.0004939694190397859, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9961534738540649, "reward_total_composite_std": 0.0004939694190397859, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1143.0} {"timestamp_utc": "2026-04-11T21:49:33Z", "mode": "train", "global_step": 1144, "epoch": 0.04417670682730924, "loss": 0.0095, "grad_norm": 5.928881645202637, "learning_rate": 6.536363636363638e-06, "num_tokens": 2467703.0, "completions/mean_length": 74.375, "completions/min_length": 71.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.375, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.09135062992572784, "rewards/meter/std": 0.11340192705392838, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.09135062992572784, "rewards/total_composite/std": 0.11340192705392838, "reward": 0.09135062992572784, "reward_std": 0.11340192705392838, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03074265830218792, "sampling/sampling_logp_difference/max": 1.1774389743804932, "sampling/importance_sampling_ratio/min": 0.3080666959285736, "sampling/importance_sampling_ratio/mean": 1.0000197887420654, "sampling/importance_sampling_ratio/max": 1.9065487384796143, "entropy": 0.14179788995534182, "clip_ratio/low_mean": 0.023183429846540093, "clip_ratio/low_min": 0.023183429846540093, "clip_ratio/high_mean": 0.008517320267856121, "clip_ratio/high_max": 0.008517320267856121, "clip_ratio/region_mean": 0.031700750114396214, "reward_total_mean": 0.09135062992572784, "reward_meter_mean": 0.09135062992572784, "reward_meter_std": 0.11340192705392838, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.09135062992572784, "reward_total_composite_std": 0.11340192705392838, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1144.0} {"timestamp_utc": "2026-04-11T21:49:37Z", "mode": "train", "global_step": 1145, "epoch": 0.04421532282978066, "loss": 0.0337, "grad_norm": 21.684476852416992, "learning_rate": 6.533333333333334e-06, "num_tokens": 2469476.0, "completions/mean_length": 51.625, "completions/min_length": 50.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.625, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9845475554466248, "rewards/meter/std": 0.006476144306361675, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9845475554466248, "rewards/total_composite/std": 0.006476144306361675, "reward": 0.9845475554466248, "reward_std": 0.006476134527474642, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024862729012966156, "sampling/sampling_logp_difference/max": 1.3360815048217773, "sampling/importance_sampling_ratio/min": 0.26287373900413513, "sampling/importance_sampling_ratio/mean": 1.0010600090026855, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10274306312203407, "clip_ratio/low_mean": 0.0048584905453026295, "clip_ratio/low_min": 0.0048584905453026295, "clip_ratio/high_mean": 0.018881766591221094, "clip_ratio/high_max": 0.018881766591221094, "clip_ratio/region_mean": 0.023740257136523724, "reward_total_mean": 0.9845475554466248, "reward_meter_mean": 0.9845475554466248, "reward_meter_std": 0.006476144306361675, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9845475554466248, "reward_total_composite_std": 0.006476144306361675, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1145.0} {"timestamp_utc": "2026-04-11T21:49:43Z", "mode": "train", "global_step": 1146, "epoch": 0.044253938832252086, "loss": 0.0004, "grad_norm": 0.23874546587467194, "learning_rate": 6.530303030303031e-06, "num_tokens": 2471677.0, "completions/mean_length": 109.125, "completions/min_length": 109.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.125, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.9984644651412964, "rewards/meter/std": 1.1474165148683824e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984644651412964, "rewards/total_composite/std": 1.1474165148683824e-05, "reward": 0.9984644651412964, "reward_std": 1.1474165148683824e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0021614141296595335, "sampling/sampling_logp_difference/max": 0.574007511138916, "sampling/importance_sampling_ratio/min": 0.750515341758728, "sampling/importance_sampling_ratio/mean": 1.0007884502410889, "sampling/importance_sampling_ratio/max": 1.7753676176071167, "entropy": 0.010223650955595076, "clip_ratio/low_mean": 0.0011363636003807187, "clip_ratio/low_min": 0.0011363636003807187, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0011363636003807187, "reward_total_mean": 0.9984644651412964, "reward_meter_mean": 0.9984644651412964, "reward_meter_std": 1.1474165148683824e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984644651412964, "reward_total_composite_std": 1.1474165148683824e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1146.0} {"timestamp_utc": "2026-04-11T21:49:47Z", "mode": "train", "global_step": 1147, "epoch": 0.04429255483472351, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.527272727272728e-06, "num_tokens": 2473461.0, "completions/mean_length": 62.0, "completions/min_length": 62.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9959487915039062, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9959487915039062, "rewards/total_composite/std": 0.0, "reward": 0.9959487915039062, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0001102359892684035, "sampling/sampling_logp_difference/max": 0.004203472752124071, "sampling/importance_sampling_ratio/min": 0.9965609908103943, "sampling/importance_sampling_ratio/mean": 1.0000803470611572, "sampling/importance_sampling_ratio/max": 1.0042123794555664, "entropy": 0.0011408074205974117, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9959487915039062, "reward_meter_mean": 0.9959487915039062, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9959487915039062, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1147.0} {"timestamp_utc": "2026-04-11T21:49:54Z", "mode": "train", "global_step": 1148, "epoch": 0.044331170837194935, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.524242424242425e-06, "num_tokens": 2477093.0, "completions/mean_length": 242.0, "completions/min_length": 242.0, "completions/max_length": 242.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 242.0, "completions/min_terminated_length": 242.0, "completions/max_terminated_length": 242.0, "rewards/meter/mean": 0.9970405697822571, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8724104762077332, "rewards/total_composite/std": 0.0, "reward": 0.8724104762077332, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 5.853495167684741e-05, "sampling/sampling_logp_difference/max": 0.01380898617208004, "sampling/importance_sampling_ratio/min": 0.9911676645278931, "sampling/importance_sampling_ratio/mean": 1.000036358833313, "sampling/importance_sampling_ratio/max": 1.0139048099517822, "entropy": 0.000526532585354289, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8724104762077332, "reward_meter_mean": 0.9970405697822571, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8724104762077332, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1148.0} {"timestamp_utc": "2026-04-11T21:50:03Z", "mode": "train", "global_step": 1149, "epoch": 0.04436978683966636, "loss": -0.0126, "grad_norm": 0.35263416171073914, "learning_rate": 6.521212121212121e-06, "num_tokens": 2481621.0, "completions/mean_length": 372.0, "completions/min_length": 370.0, "completions/max_length": 386.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 372.0, "completions/min_terminated_length": 370.0, "completions/max_terminated_length": 386.0, "rewards/meter/mean": 0.9970118999481201, "rewards/meter/std": 1.4546116062774672e-06, "rewards/count_adherence/mean": 0.9270833730697632, "rewards/count_adherence/std": 0.029462777078151703, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9243130683898926, "rewards/total_composite/std": 0.029373299330472946, "reward": 0.9243130683898926, "reward_std": 0.029373306781053543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0017782310023903847, "sampling/sampling_logp_difference/max": 2.289862632751465, "sampling/importance_sampling_ratio/min": 0.10128037631511688, "sampling/importance_sampling_ratio/mean": 0.99934321641922, "sampling/importance_sampling_ratio/max": 1.0790009498596191, "entropy": 0.0008104511016426841, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.00032383418874815106, "clip_ratio/high_max": 0.00032383418874815106, "clip_ratio/region_mean": 0.00032383418874815106, "reward_total_mean": 0.9243130683898926, "reward_meter_mean": 0.9970118999481201, "reward_meter_std": 1.4546116062774672e-06, "reward_count_adherence_mean": 0.9270833730697632, "reward_count_adherence_std": 0.029462777078151703, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9243130683898926, "reward_total_composite_std": 0.029373299330472946, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1149.0} {"timestamp_utc": "2026-04-11T21:50:08Z", "mode": "train", "global_step": 1150, "epoch": 0.04440840284213778, "loss": 0.0522, "grad_norm": 5.2525129318237305, "learning_rate": 6.5181818181818195e-06, "num_tokens": 2483289.0, "completions/mean_length": 57.5, "completions/min_length": 54.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8777114748954773, "rewards/meter/std": 0.16049861907958984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8777114748954773, "rewards/total_composite/std": 0.16049861907958984, "reward": 0.8777114748954773, "reward_std": 0.16049860417842865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024218173697590828, "sampling/sampling_logp_difference/max": 5.00484561920166, "sampling/importance_sampling_ratio/min": 0.006705376319587231, "sampling/importance_sampling_ratio/mean": 0.9977035522460938, "sampling/importance_sampling_ratio/max": 1.5273561477661133, "entropy": 0.05149026960134506, "clip_ratio/low_mean": 0.004067460540682077, "clip_ratio/low_min": 0.004067460540682077, "clip_ratio/high_mean": 0.009015594609081745, "clip_ratio/high_max": 0.009015594609081745, "clip_ratio/region_mean": 0.013083055149763823, "reward_total_mean": 0.8777114748954773, "reward_meter_mean": 0.8777114748954773, "reward_meter_std": 0.16049861907958984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8777114748954773, "reward_total_composite_std": 0.16049861907958984, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1150.0} {"timestamp_utc": "2026-04-11T21:51:32Z", "mode": "eval", "global_step": 1150, "epoch": 0.04440840284213778, "eval_loss": NaN, "eval_runtime": 84.6125, "eval_samples_per_second": 1.229, "eval_steps_per_second": 0.154, "eval_num_tokens": 2483289.0, "eval_completions/mean_length": 216.47115384615384, "eval_completions/min_length": 60.76923076923077, "eval_completions/max_length": 447.0, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/mean_terminated_length": 197.60806157038763, "eval_completions/min_terminated_length": 60.76923076923077, "eval_completions/max_terminated_length": 396.38461538461536, "eval_rewards/meter/mean": 0.5331913347427661, "eval_rewards/meter/std": 0.45816060442190903, "eval_rewards/count_adherence/mean": 0.9469390053015488, "eval_rewards/count_adherence/std": 0.06932907207654072, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/total_composite/mean": 0.511953374514213, "eval_rewards/total_composite/std": 0.440208015533594, "eval_reward": 0.511953374514213, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0024267814587801695, "eval_sampling/sampling_logp_difference/max": 0.5305701494216919, "eval_sampling/importance_sampling_ratio/min": 0.629830559858909, "eval_sampling/importance_sampling_ratio/mean": 1.0007386207580566, "eval_sampling/importance_sampling_ratio/max": 1.403554081916809, "eval_entropy": 0.017647026679836787, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.511953374514213, "eval_reward_meter_mean": 0.5331913347427661, "eval_reward_meter_std": 0.45816060442190903, "eval_reward_count_adherence_mean": 0.9469390053015488, "eval_reward_count_adherence_std": 0.06932907207654072, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_total_composite_mean": 0.511953374514213, "eval_reward_total_composite_std": 0.440208015533594, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1150.0} {"timestamp_utc": "2026-04-11T21:51:41Z", "mode": "train", "global_step": 1151, "epoch": 0.04444701884460921, "loss": 0.0504, "grad_norm": 3.7881219387054443, "learning_rate": 6.515151515151516e-06, "num_tokens": 2485639.0, "completions/mean_length": 122.75, "completions/min_length": 120.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.75, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.888770341873169, "rewards/meter/std": 0.21289391815662384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.888770341873169, "rewards/total_composite/std": 0.21289391815662384, "reward": 0.888770341873169, "reward_std": 0.21289391815662384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020240498706698418, "sampling/sampling_logp_difference/max": 6.1473388671875, "sampling/importance_sampling_ratio/min": 0.0021391669288277626, "sampling/importance_sampling_ratio/mean": 0.9987115263938904, "sampling/importance_sampling_ratio/max": 1.9864379167556763, "entropy": 0.013336000498384237, "clip_ratio/low_mean": 0.003968254197388887, "clip_ratio/low_min": 0.003968254197388887, "clip_ratio/high_mean": 0.006250000325962901, "clip_ratio/high_max": 0.006250000325962901, "clip_ratio/region_mean": 0.010218254523351789, "reward_total_mean": 0.888770341873169, "reward_meter_mean": 0.888770341873169, "reward_meter_std": 0.21289391815662384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.888770341873169, "reward_total_composite_std": 0.21289391815662384, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1151.0} {"timestamp_utc": "2026-04-11T21:51:45Z", "mode": "train", "global_step": 1152, "epoch": 0.04448563484708063, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.512121212121213e-06, "num_tokens": 2487423.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 9.262973617296666e-05, "sampling/sampling_logp_difference/max": 0.002871689386665821, "sampling/importance_sampling_ratio/min": 0.9996728897094727, "sampling/importance_sampling_ratio/mean": 1.0000879764556885, "sampling/importance_sampling_ratio/max": 1.002875804901123, "entropy": 0.0009597428434062749, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1152.0} {"timestamp_utc": "2026-04-11T21:51:50Z", "mode": "train", "global_step": 1153, "epoch": 0.044524250849552055, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.5090909090909095e-06, "num_tokens": 2489239.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9972342252731323, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9972342252731323, "rewards/total_composite/std": 0.0, "reward": 0.9972342252731323, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00015291250019799918, "sampling/sampling_logp_difference/max": 0.014015945605933666, "sampling/importance_sampling_ratio/min": 0.9860818386077881, "sampling/importance_sampling_ratio/mean": 1.0000462532043457, "sampling/importance_sampling_ratio/max": 1.0045279264450073, "entropy": 0.00251059714355506, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9972342252731323, "reward_meter_mean": 0.9972342252731323, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9972342252731323, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1153.0} {"timestamp_utc": "2026-04-11T21:51:55Z", "mode": "train", "global_step": 1154, "epoch": 0.04456286685202348, "loss": 0.096, "grad_norm": 5.494170665740967, "learning_rate": 6.506060606060607e-06, "num_tokens": 2491171.0, "completions/mean_length": 65.5, "completions/min_length": 53.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.13582952320575714, "rewards/meter/std": 0.19803810119628906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.13582952320575714, "rewards/total_composite/std": 0.19803810119628906, "reward": 0.13582952320575714, "reward_std": 0.19803808629512787, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041303541511297226, "sampling/sampling_logp_difference/max": 4.165966033935547, "sampling/importance_sampling_ratio/min": 0.015514720231294632, "sampling/importance_sampling_ratio/mean": 0.9954270124435425, "sampling/importance_sampling_ratio/max": 1.913017988204956, "entropy": 0.0860102130100131, "clip_ratio/low_mean": 0.013047155574895442, "clip_ratio/low_min": 0.013047155574895442, "clip_ratio/high_mean": 0.008954269345849752, "clip_ratio/high_max": 0.008954269345849752, "clip_ratio/region_mean": 0.022001424920745194, "reward_total_mean": 0.13582952320575714, "reward_meter_mean": 0.13582952320575714, "reward_meter_std": 0.19803810119628906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.13582952320575714, "reward_total_composite_std": 0.19803810119628906, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1154.0} {"timestamp_utc": "2026-04-11T21:52:02Z", "mode": "train", "global_step": 1155, "epoch": 0.0446014828544949, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.503030303030303e-06, "num_tokens": 2493723.0, "completions/mean_length": 130.0, "completions/min_length": 130.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.0, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9970986843109131, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9970986843109131, "rewards/total_composite/std": 0.0, "reward": 0.9970986843109131, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0001729379582684487, "sampling/sampling_logp_difference/max": 0.117275670170784, "sampling/importance_sampling_ratio/min": 0.8893400430679321, "sampling/importance_sampling_ratio/mean": 0.9999405145645142, "sampling/importance_sampling_ratio/max": 1.0045697689056396, "entropy": 0.0005436375031422358, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9970986843109131, "reward_meter_mean": 0.9970986843109131, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9970986843109131, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1155.0} {"timestamp_utc": "2026-04-11T21:52:12Z", "mode": "train", "global_step": 1156, "epoch": 0.04464009885696633, "loss": -0.0346, "grad_norm": 3.3614652156829834, "learning_rate": 6.5000000000000004e-06, "num_tokens": 2497463.0, "completions/mean_length": 304.5, "completions/min_length": 256.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 274.8571472167969, "completions/min_terminated_length": 256.0, "completions/max_terminated_length": 288.0, "rewards/meter/mean": 0.028450578451156616, "rewards/meter/std": 0.04791045933961868, "rewards/count_adherence/mean": 0.9285714626312256, "rewards/count_adherence/std": 0.2020305097103119, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.025792036205530167, "rewards/total_composite/std": 0.048944856971502304, "reward": 0.025792036205530167, "reward_std": 0.048944856971502304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024456597864627838, "sampling/sampling_logp_difference/max": 4.192187309265137, "sampling/importance_sampling_ratio/min": 0.015113191679120064, "sampling/importance_sampling_ratio/mean": 1.001096487045288, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04777472233399749, "clip_ratio/low_mean": 0.014450150192715228, "clip_ratio/low_min": 0.014450150192715228, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.016403275192715228, "reward_total_mean": 0.025792036205530167, "reward_meter_mean": 0.028450578451156616, "reward_meter_std": 0.04791045933961868, "reward_count_adherence_mean": 0.9285714626312256, "reward_count_adherence_std": 0.2020305097103119, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.025792036205530167, "reward_total_composite_std": 0.048944856971502304, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1156.0} {"timestamp_utc": "2026-04-11T21:52:16Z", "mode": "train", "global_step": 1157, "epoch": 0.04467871485943775, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.496969696969697e-06, "num_tokens": 2499215.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9972342252731323, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9972342252731323, "rewards/total_composite/std": 0.0, "reward": 0.9972342252731323, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0148831931874156, "sampling/sampling_logp_difference/max": 3.352174758911133, "sampling/importance_sampling_ratio/min": 0.03500813990831375, "sampling/importance_sampling_ratio/mean": 0.9938147664070129, "sampling/importance_sampling_ratio/max": 1.0112634897232056, "entropy": 0.006829831196228042, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9972342252731323, "reward_meter_mean": 0.9972342252731323, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9972342252731323, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1157.0} {"timestamp_utc": "2026-04-11T21:52:21Z", "mode": "train", "global_step": 1158, "epoch": 0.044717330861909176, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.493939393939395e-06, "num_tokens": 2501127.0, "completions/mean_length": 72.0, "completions/min_length": 72.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9888595938682556, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9888595938682556, "rewards/total_composite/std": 0.0, "reward": 0.9888595938682556, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00017435323388781399, "sampling/sampling_logp_difference/max": 0.0043810089118778706, "sampling/importance_sampling_ratio/min": 0.9968689680099487, "sampling/importance_sampling_ratio/mean": 1.000144600868225, "sampling/importance_sampling_ratio/max": 1.0043905973434448, "entropy": 0.0018330513557884842, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9888595938682556, "reward_meter_mean": 0.9888595938682556, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9888595938682556, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1158.0} {"timestamp_utc": "2026-04-11T21:52:26Z", "mode": "train", "global_step": 1159, "epoch": 0.0447559468643806, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.490909090909091e-06, "num_tokens": 2503135.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00013290751667227596, "sampling/sampling_logp_difference/max": 0.006528853438794613, "sampling/importance_sampling_ratio/min": 0.9994356036186218, "sampling/importance_sampling_ratio/mean": 1.0001311302185059, "sampling/importance_sampling_ratio/max": 1.0065501928329468, "entropy": 0.0013790328812319785, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1159.0} {"timestamp_utc": "2026-04-11T21:52:31Z", "mode": "train", "global_step": 1160, "epoch": 0.044794562866852024, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.487878787878789e-06, "num_tokens": 2504759.0, "completions/mean_length": 62.0, "completions/min_length": 62.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9959487915039062, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9959487915039062, "rewards/total_composite/std": 0.0, "reward": 0.9959487915039062, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0001226484455401078, "sampling/sampling_logp_difference/max": 0.005975149571895599, "sampling/importance_sampling_ratio/min": 0.9942424893379211, "sampling/importance_sampling_ratio/mean": 1.0000890493392944, "sampling/importance_sampling_ratio/max": 1.005993127822876, "entropy": 0.001158960752945859, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9959487915039062, "reward_meter_mean": 0.9959487915039062, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9959487915039062, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1160.0} {"timestamp_utc": "2026-04-11T21:52:36Z", "mode": "train", "global_step": 1161, "epoch": 0.04483317886932345, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.484848484848485e-06, "num_tokens": 2506847.0, "completions/mean_length": 96.0, "completions/min_length": 96.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9888964295387268, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9888964295387268, "rewards/total_composite/std": 0.0, "reward": 0.9888964295387268, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00014731692499481142, "sampling/sampling_logp_difference/max": 0.005519423168152571, "sampling/importance_sampling_ratio/min": 0.9944958090782166, "sampling/importance_sampling_ratio/mean": 1.0000895261764526, "sampling/importance_sampling_ratio/max": 1.0041718482971191, "entropy": 0.0013596047574537806, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9888964295387268, "reward_meter_mean": 0.9888964295387268, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9888964295387268, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1161.0} {"timestamp_utc": "2026-04-11T21:52:41Z", "mode": "train", "global_step": 1162, "epoch": 0.04487179487179487, "loss": -0.0067, "grad_norm": 9.667710304260254, "learning_rate": 6.481818181818182e-06, "num_tokens": 2508562.0, "completions/mean_length": 54.375, "completions/min_length": 53.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9774544835090637, "rewards/meter/std": 0.007204634603112936, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9774544835090637, "rewards/total_composite/std": 0.007204634603112936, "reward": 0.9774544835090637, "reward_std": 0.007204628549516201, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024649647995829582, "sampling/sampling_logp_difference/max": 3.8922629356384277, "sampling/importance_sampling_ratio/min": 0.02039913274347782, "sampling/importance_sampling_ratio/mean": 0.9978480339050293, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.028170868754386902, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/high_mean": 0.004590633558109403, "clip_ratio/high_max": 0.004590633558109403, "clip_ratio/region_mean": 0.009220263222232461, "reward_total_mean": 0.9774544835090637, "reward_meter_mean": 0.9774544835090637, "reward_meter_std": 0.007204634603112936, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9774544835090637, "reward_total_composite_std": 0.007204634603112936, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1162.0} {"timestamp_utc": "2026-04-11T21:52:47Z", "mode": "train", "global_step": 1163, "epoch": 0.044910410874266296, "loss": -0.0003, "grad_norm": 0.0491514727473259, "learning_rate": 6.478787878787879e-06, "num_tokens": 2511042.0, "completions/mean_length": 145.0, "completions/min_length": 145.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.0, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9984663724899292, "rewards/meter/std": 3.918135462299688e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984663724899292, "rewards/total_composite/std": 3.918135462299688e-06, "reward": 0.9984663724899292, "reward_std": 3.918005404557334e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0012445234460756183, "sampling/sampling_logp_difference/max": 0.6710500717163086, "sampling/importance_sampling_ratio/min": 0.8294064402580261, "sampling/importance_sampling_ratio/mean": 1.000813603401184, "sampling/importance_sampling_ratio/max": 1.9562904834747314, "entropy": 0.006222540338058025, "clip_ratio/low_mean": 0.0008620689623057842, "clip_ratio/low_min": 0.0008620689623057842, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0008620689623057842, "reward_total_mean": 0.9984663724899292, "reward_meter_mean": 0.9984663724899292, "reward_meter_std": 3.918135462299688e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984663724899292, "reward_total_composite_std": 3.918135462299688e-06, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1163.0} {"timestamp_utc": "2026-04-11T21:52:51Z", "mode": "train", "global_step": 1164, "epoch": 0.04494902687673772, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.475757575757576e-06, "num_tokens": 2512906.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 8.520409755874425e-05, "sampling/sampling_logp_difference/max": 0.0016170135932043195, "sampling/importance_sampling_ratio/min": 0.9988471865653992, "sampling/importance_sampling_ratio/mean": 1.000075340270996, "sampling/importance_sampling_ratio/max": 1.0016183853149414, "entropy": 0.0006748539672116749, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1164.0} {"timestamp_utc": "2026-04-11T21:52:56Z", "mode": "train", "global_step": 1165, "epoch": 0.044987642879209144, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.472727272727272e-06, "num_tokens": 2514546.0, "completions/mean_length": 62.0, "completions/min_length": 62.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9959487915039062, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9959487915039062, "rewards/total_composite/std": 0.0, "reward": 0.9959487915039062, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00018705571710597724, "sampling/sampling_logp_difference/max": 0.003838915377855301, "sampling/importance_sampling_ratio/min": 0.996885359287262, "sampling/importance_sampling_ratio/mean": 1.0001657009124756, "sampling/importance_sampling_ratio/max": 1.003846287727356, "entropy": 0.0014549151237588376, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9959487915039062, "reward_meter_mean": 0.9959487915039062, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9959487915039062, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1165.0} {"timestamp_utc": "2026-04-11T21:53:01Z", "mode": "train", "global_step": 1166, "epoch": 0.04502625888168057, "loss": 0.0454, "grad_norm": 190.9034881591797, "learning_rate": 6.4696969696969705e-06, "num_tokens": 2516434.0, "completions/mean_length": 58.0, "completions/min_length": 55.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.996626615524292, "rewards/meter/std": 0.001510909991338849, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.996626615524292, "rewards/total_composite/std": 0.001510909991338849, "reward": 0.996626615524292, "reward_std": 0.0015109025407582521, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025698505342006683, "sampling/sampling_logp_difference/max": 1.1933188438415527, "sampling/importance_sampling_ratio/min": 0.3032132685184479, "sampling/importance_sampling_ratio/mean": 0.9995715618133545, "sampling/importance_sampling_ratio/max": 1.7860766649246216, "entropy": 0.059585667215287685, "clip_ratio/low_mean": 0.006657268386334181, "clip_ratio/low_min": 0.006657268386334181, "clip_ratio/high_mean": 0.008513708598911762, "clip_ratio/high_max": 0.008513708598911762, "clip_ratio/region_mean": 0.015170976985245943, "reward_total_mean": 0.996626615524292, "reward_meter_mean": 0.996626615524292, "reward_meter_std": 0.001510909991338849, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.996626615524292, "reward_total_composite_std": 0.001510909991338849, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1166.0} {"timestamp_utc": "2026-04-11T21:53:06Z", "mode": "train", "global_step": 1167, "epoch": 0.04506487488415199, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.466666666666667e-06, "num_tokens": 2518002.0, "completions/mean_length": 48.0, "completions/min_length": 48.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9887858629226685, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9887858629226685, "rewards/total_composite/std": 0.0, "reward": 0.9887858629226685, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0013065863167867064, "sampling/sampling_logp_difference/max": 0.18986795842647552, "sampling/importance_sampling_ratio/min": 0.8270683288574219, "sampling/importance_sampling_ratio/mean": 0.9993685483932495, "sampling/importance_sampling_ratio/max": 1.032031774520874, "entropy": 0.006112938834121451, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9887858629226685, "reward_meter_mean": 0.9887858629226685, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9887858629226685, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1167.0} {"timestamp_utc": "2026-04-11T21:53:11Z", "mode": "train", "global_step": 1168, "epoch": 0.04510349088662342, "loss": 0.0287, "grad_norm": 7.1253252029418945, "learning_rate": 6.463636363636364e-06, "num_tokens": 2520049.0, "completions/mean_length": 81.875, "completions/min_length": 80.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.875, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.03959798067808151, "rewards/meter/std": 0.011728010140359402, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.038292575627565384, "rewards/total_composite/std": 0.01326083205640316, "reward": 0.038292575627565384, "reward_std": 0.01326083205640316, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028142500668764114, "sampling/sampling_logp_difference/max": 4.7991814613342285, "sampling/importance_sampling_ratio/min": 0.008236486464738846, "sampling/importance_sampling_ratio/mean": 0.9953240752220154, "sampling/importance_sampling_ratio/max": 1.6731361150741577, "entropy": 0.03918551583774388, "clip_ratio/low_mean": 0.014004629920236766, "clip_ratio/low_min": 0.014004629920236766, "clip_ratio/high_mean": 0.004611280397512019, "clip_ratio/high_max": 0.004611280397512019, "clip_ratio/region_mean": 0.018615910317748785, "reward_total_mean": 0.038292575627565384, "reward_meter_mean": 0.03959798067808151, "reward_meter_std": 0.011728010140359402, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.038292575627565384, "reward_total_composite_std": 0.01326083205640316, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1168.0} {"timestamp_utc": "2026-04-11T21:53:15Z", "mode": "train", "global_step": 1169, "epoch": 0.04514210688909484, "loss": -0.044, "grad_norm": 4.53950834274292, "learning_rate": 6.460606060606061e-06, "num_tokens": 2521727.0, "completions/mean_length": 58.75, "completions/min_length": 55.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9976530075073242, "rewards/meter/std": 0.0003900358860846609, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9976530075073242, "rewards/total_composite/std": 0.0003900358860846609, "reward": 0.9976530075073242, "reward_std": 0.00039003457641229033, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013675752095878124, "sampling/sampling_logp_difference/max": 0.7795699834823608, "sampling/importance_sampling_ratio/min": 0.4586032032966614, "sampling/importance_sampling_ratio/mean": 0.9979230165481567, "sampling/importance_sampling_ratio/max": 1.4813544750213623, "entropy": 0.0419846111908555, "clip_ratio/low_mean": 0.006818181602284312, "clip_ratio/low_min": 0.006818181602284312, "clip_ratio/high_mean": 0.002016128972172737, "clip_ratio/high_max": 0.002016128972172737, "clip_ratio/region_mean": 0.00883431057445705, "reward_total_mean": 0.9976530075073242, "reward_meter_mean": 0.9976530075073242, "reward_meter_std": 0.0003900358860846609, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9976530075073242, "reward_total_composite_std": 0.0003900358860846609, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1169.0} {"timestamp_utc": "2026-04-11T21:53:26Z", "mode": "train", "global_step": 1170, "epoch": 0.045180722891566265, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.457575757575758e-06, "num_tokens": 2523215.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9969618916511536, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8545387387275696, "rewards/total_composite/std": 0.0, "reward": 0.8545387387275696, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8545387387275696, "reward_meter_mean": 0.9969618916511536, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8545387387275696, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1170.0} {"timestamp_utc": "2026-04-11T21:53:30Z", "mode": "train", "global_step": 1171, "epoch": 0.04521933889403769, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.454545454545456e-06, "num_tokens": 2524991.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00013418152229860425, "sampling/sampling_logp_difference/max": 0.004571585915982723, "sampling/importance_sampling_ratio/min": 0.999534547328949, "sampling/importance_sampling_ratio/mean": 1.0001263618469238, "sampling/importance_sampling_ratio/max": 1.0045820474624634, "entropy": 0.0013301519793458283, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1171.0} {"timestamp_utc": "2026-04-11T21:53:36Z", "mode": "train", "global_step": 1172, "epoch": 0.04525795489650911, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.451515151515152e-06, "num_tokens": 2527167.0, "completions/mean_length": 96.0, "completions/min_length": 96.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9888964295387268, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9888964295387268, "rewards/total_composite/std": 0.0, "reward": 0.9888964295387268, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002919211983680725, "sampling/sampling_logp_difference/max": 0.04608858376741409, "sampling/importance_sampling_ratio/min": 0.9549573659896851, "sampling/importance_sampling_ratio/mean": 1.00016450881958, "sampling/importance_sampling_ratio/max": 1.0190335512161255, "entropy": 0.00159282027016161, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9888964295387268, "reward_meter_mean": 0.9888964295387268, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9888964295387268, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1172.0} {"timestamp_utc": "2026-04-11T21:53:41Z", "mode": "train", "global_step": 1173, "epoch": 0.04529657089898054, "loss": 0.0337, "grad_norm": 20.588041305541992, "learning_rate": 6.4484848484848496e-06, "num_tokens": 2528827.0, "completions/mean_length": 48.5, "completions/min_length": 48.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9464411735534668, "rewards/meter/std": 0.11976895481348038, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9464411735534668, "rewards/total_composite/std": 0.11976895481348038, "reward": 0.9464411735534668, "reward_std": 0.11976895481348038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008157458156347275, "sampling/sampling_logp_difference/max": 1.4072527885437012, "sampling/importance_sampling_ratio/min": 0.244814932346344, "sampling/importance_sampling_ratio/mean": 1.0024529695510864, "sampling/importance_sampling_ratio/max": 1.9219340085983276, "entropy": 0.018341065326239914, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/high_mean": 0.0026041667442768812, "clip_ratio/high_max": 0.0026041667442768812, "clip_ratio/region_mean": 0.007411859231069684, "reward_total_mean": 0.9464411735534668, "reward_meter_mean": 0.9464411735534668, "reward_meter_std": 0.11976895481348038, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9464411735534668, "reward_total_composite_std": 0.11976895481348038, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1173.0} {"timestamp_utc": "2026-04-11T21:53:45Z", "mode": "train", "global_step": 1174, "epoch": 0.04533518690145196, "loss": 0.032, "grad_norm": 12.414958953857422, "learning_rate": 6.445454545454546e-06, "num_tokens": 2530280.0, "completions/mean_length": 30.625, "completions/min_length": 30.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9770581126213074, "rewards/meter/std": 0.03885127976536751, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9770581126213074, "rewards/total_composite/std": 0.03885127976536751, "reward": 0.9770581126213074, "reward_std": 0.03885127976536751, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00510073360055685, "sampling/sampling_logp_difference/max": 0.34801721572875977, "sampling/importance_sampling_ratio/min": 0.7204657793045044, "sampling/importance_sampling_ratio/mean": 1.0016231536865234, "sampling/importance_sampling_ratio/max": 1.4162566661834717, "entropy": 0.015783087466843426, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004166666883975267, "clip_ratio/high_max": 0.004166666883975267, "clip_ratio/region_mean": 0.004166666883975267, "reward_total_mean": 0.9770581126213074, "reward_meter_mean": 0.9770581126213074, "reward_meter_std": 0.03885127976536751, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9770581126213074, "reward_total_composite_std": 0.03885127976536751, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1174.0} {"timestamp_utc": "2026-04-11T21:53:50Z", "mode": "train", "global_step": 1175, "epoch": 0.045373802903923385, "loss": 0.0157, "grad_norm": 10.53234577178955, "learning_rate": 6.442424242424243e-06, "num_tokens": 2531910.0, "completions/mean_length": 62.75, "completions/min_length": 62.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.98736572265625, "rewards/meter/std": 0.02987661585211754, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.98736572265625, "rewards/total_composite/std": 0.02987661585211754, "reward": 0.98736572265625, "reward_std": 0.02987661026418209, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009396832436323166, "sampling/sampling_logp_difference/max": 1.0190973281860352, "sampling/importance_sampling_ratio/min": 0.36092060804367065, "sampling/importance_sampling_ratio/mean": 1.0028626918792725, "sampling/importance_sampling_ratio/max": 1.5600427389144897, "entropy": 0.03662474290467799, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/high_mean": 0.0059843831695616245, "clip_ratio/high_max": 0.0059843831695616245, "clip_ratio/region_mean": 0.007937508169561625, "reward_total_mean": 0.98736572265625, "reward_meter_mean": 0.98736572265625, "reward_meter_std": 0.02987661585211754, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.98736572265625, "reward_total_composite_std": 0.02987661585211754, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1175.0} {"timestamp_utc": "2026-04-11T21:53:58Z", "mode": "train", "global_step": 1176, "epoch": 0.04541241890639481, "loss": 0.0004, "grad_norm": 0.008205018006265163, "learning_rate": 6.43939393939394e-06, "num_tokens": 2536102.0, "completions/mean_length": 307.0, "completions/min_length": 307.0, "completions/max_length": 307.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 307.0, "completions/min_terminated_length": 307.0, "completions/max_terminated_length": 307.0, "rewards/meter/mean": 0.9984632730484009, "rewards/meter/std": 1.9595856883825036e-06, "rewards/count_adherence/mean": 0.8888888955116272, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8875229358673096, "rewards/total_composite/std": 1.7389269260092988e-06, "reward": 0.8875229358673096, "reward_std": 1.7291217773163226e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00040480130701325834, "sampling/sampling_logp_difference/max": 0.38402700424194336, "sampling/importance_sampling_ratio/min": 0.681113064289093, "sampling/importance_sampling_ratio/mean": 0.9998529553413391, "sampling/importance_sampling_ratio/max": 1.0504670143127441, "entropy": 0.001436470149201341, "clip_ratio/low_mean": 0.0004071661096531898, "clip_ratio/low_min": 0.0004071661096531898, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0004071661096531898, "reward_total_mean": 0.8875229358673096, "reward_meter_mean": 0.9984632730484009, "reward_meter_std": 1.9595856883825036e-06, "reward_count_adherence_mean": 0.8888888955116272, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8875229358673096, "reward_total_composite_std": 1.7389269260092988e-06, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1176.0} {"timestamp_utc": "2026-04-11T21:54:03Z", "mode": "train", "global_step": 1177, "epoch": 0.045451034908866234, "loss": -0.0054, "grad_norm": 3.3288590908050537, "learning_rate": 6.436363636363637e-06, "num_tokens": 2537728.0, "completions/mean_length": 56.25, "completions/min_length": 56.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9876357316970825, "rewards/meter/std": 0.00018677377374842763, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9876357316970825, "rewards/total_composite/std": 0.00018677377374842763, "reward": 0.9876357316970825, "reward_std": 0.00018675869796425104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014386294409632683, "sampling/sampling_logp_difference/max": 2.6926071643829346, "sampling/importance_sampling_ratio/min": 0.0677042007446289, "sampling/importance_sampling_ratio/mean": 0.9979600310325623, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.014292308245785534, "clip_ratio/low_mean": 0.0022321429569274187, "clip_ratio/low_min": 0.0022321429569274187, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0022321429569274187, "reward_total_mean": 0.9876357316970825, "reward_meter_mean": 0.9876357316970825, "reward_meter_std": 0.00018677377374842763, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9876357316970825, "reward_total_composite_std": 0.00018677377374842763, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1177.0} {"timestamp_utc": "2026-04-11T21:54:08Z", "mode": "train", "global_step": 1178, "epoch": 0.04548965091133766, "loss": 0.0171, "grad_norm": 13.329391479492188, "learning_rate": 6.433333333333333e-06, "num_tokens": 2539308.0, "completions/mean_length": 48.5, "completions/min_length": 47.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9343859553337097, "rewards/meter/std": 0.11850643157958984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9343859553337097, "rewards/total_composite/std": 0.11850643157958984, "reward": 0.9343859553337097, "reward_std": 0.11850643903017044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027483860030770302, "sampling/sampling_logp_difference/max": 1.6382932662963867, "sampling/importance_sampling_ratio/min": 0.19431141018867493, "sampling/importance_sampling_ratio/mean": 0.9949454069137573, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0684116561897099, "clip_ratio/low_mean": 0.0049019609577953815, "clip_ratio/low_min": 0.0049019609577953815, "clip_ratio/high_mean": 0.01535926922224462, "clip_ratio/high_max": 0.01535926922224462, "clip_ratio/region_mean": 0.020261230180040002, "reward_total_mean": 0.9343859553337097, "reward_meter_mean": 0.9343859553337097, "reward_meter_std": 0.11850643157958984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9343859553337097, "reward_total_composite_std": 0.11850643157958984, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1178.0} {"timestamp_utc": "2026-04-11T21:54:14Z", "mode": "train", "global_step": 1179, "epoch": 0.04552826691380908, "loss": -0.0044, "grad_norm": 4.12630033493042, "learning_rate": 6.430303030303031e-06, "num_tokens": 2541097.0, "completions/mean_length": 61.625, "completions/min_length": 61.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9966377019882202, "rewards/meter/std": 0.0008516657399013638, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9966377019882202, "rewards/total_composite/std": 0.0008516657399013638, "reward": 0.9966377019882202, "reward_std": 0.0008516703965142369, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012902844697237015, "sampling/sampling_logp_difference/max": 1.0336610078811646, "sampling/importance_sampling_ratio/min": 0.35570234060287476, "sampling/importance_sampling_ratio/mean": 0.995635449886322, "sampling/importance_sampling_ratio/max": 1.394187092781067, "entropy": 0.03527964395470917, "clip_ratio/low_mean": 0.008163669612258673, "clip_ratio/low_min": 0.008163669612258673, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/region_mean": 0.010147796710953116, "reward_total_mean": 0.9966377019882202, "reward_meter_mean": 0.9966377019882202, "reward_meter_std": 0.0008516657399013638, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9966377019882202, "reward_total_composite_std": 0.0008516657399013638, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1179.0} {"timestamp_utc": "2026-04-11T21:54:20Z", "mode": "train", "global_step": 1180, "epoch": 0.045566882916280506, "loss": -0.0, "grad_norm": 0.15215128660202026, "learning_rate": 6.427272727272728e-06, "num_tokens": 2544066.0, "completions/mean_length": 162.125, "completions/min_length": 162.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 162.125, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.9969795942306519, "rewards/meter/std": 5.755153688369319e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9969795942306519, "rewards/total_composite/std": 5.755153688369319e-05, "reward": 0.9969795942306519, "reward_std": 5.756055543315597e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00256105768494308, "sampling/sampling_logp_difference/max": 1.255385160446167, "sampling/importance_sampling_ratio/min": 0.28496605157852173, "sampling/importance_sampling_ratio/mean": 0.9987585544586182, "sampling/importance_sampling_ratio/max": 1.2257132530212402, "entropy": 0.005278411292238161, "clip_ratio/low_mean": 0.0015432098880410194, "clip_ratio/low_min": 0.0015432098880410194, "clip_ratio/high_mean": 0.0007668711477890611, "clip_ratio/high_max": 0.0007668711477890611, "clip_ratio/region_mean": 0.0023100810358300805, "reward_total_mean": 0.9969795942306519, "reward_meter_mean": 0.9969795942306519, "reward_meter_std": 5.755153688369319e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9969795942306519, "reward_total_composite_std": 5.755153688369319e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1180.0} {"timestamp_utc": "2026-04-11T21:54:25Z", "mode": "train", "global_step": 1181, "epoch": 0.04560549891875193, "loss": -0.0015, "grad_norm": 1.7918148040771484, "learning_rate": 6.424242424242425e-06, "num_tokens": 2545862.0, "completions/mean_length": 62.5, "completions/min_length": 62.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.998000979423523, "rewards/meter/std": 5.412446989794262e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.998000979423523, "rewards/total_composite/std": 5.412446989794262e-05, "reward": 0.998000979423523, "reward_std": 5.414646147983149e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008854641579091549, "sampling/sampling_logp_difference/max": 0.7983036041259766, "sampling/importance_sampling_ratio/min": 0.4500918388366699, "sampling/importance_sampling_ratio/mean": 0.9972570538520813, "sampling/importance_sampling_ratio/max": 1.2694493532180786, "entropy": 0.02104826516006142, "clip_ratio/low_mean": 0.002016128972172737, "clip_ratio/low_min": 0.002016128972172737, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/region_mean": 0.004000256070867181, "reward_total_mean": 0.998000979423523, "reward_meter_mean": 0.998000979423523, "reward_meter_std": 5.412446989794262e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.998000979423523, "reward_total_composite_std": 5.412446989794262e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1181.0} {"timestamp_utc": "2026-04-11T21:54:31Z", "mode": "train", "global_step": 1182, "epoch": 0.045644114921223354, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.4212121212121215e-06, "num_tokens": 2548958.0, "completions/mean_length": 210.0, "completions/min_length": 210.0, "completions/max_length": 210.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 210.0, "completions/min_terminated_length": 210.0, "completions/max_terminated_length": 210.0, "rewards/meter/mean": 0.9969598650932312, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8545370101928711, "rewards/total_composite/std": 0.0, "reward": 0.8545370101928711, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0003782480489462614, "sampling/sampling_logp_difference/max": 0.14583073556423187, "sampling/importance_sampling_ratio/min": 0.8643040657043457, "sampling/importance_sampling_ratio/mean": 0.9999819397926331, "sampling/importance_sampling_ratio/max": 1.0694695711135864, "entropy": 0.0029908385331509635, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8545370101928711, "reward_meter_mean": 0.9969598650932312, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8545370101928711, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1182.0} {"timestamp_utc": "2026-04-11T21:54:36Z", "mode": "train", "global_step": 1183, "epoch": 0.04568273092369478, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.418181818181819e-06, "num_tokens": 2550854.0, "completions/mean_length": 73.0, "completions/min_length": 73.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9984769821166992, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984769821166992, "rewards/total_composite/std": 0.0, "reward": 0.9984769821166992, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0008001961396075785, "sampling/sampling_logp_difference/max": 0.0588974803686142, "sampling/importance_sampling_ratio/min": 0.9428033828735352, "sampling/importance_sampling_ratio/mean": 1.0005358457565308, "sampling/importance_sampling_ratio/max": 1.0240943431854248, "entropy": 0.007733863138128072, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9984769821166992, "reward_meter_mean": 0.9984769821166992, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984769821166992, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1183.0} {"timestamp_utc": "2026-04-11T21:54:42Z", "mode": "train", "global_step": 1184, "epoch": 0.0457213469261662, "loss": 0.0258, "grad_norm": 5.115583896636963, "learning_rate": 6.415151515151515e-06, "num_tokens": 2553118.0, "completions/mean_length": 118.0, "completions/min_length": 107.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.0, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.011638942174613476, "rewards/meter/std": 0.01774718053638935, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.011638942174613476, "rewards/total_composite/std": 0.01774718053638935, "reward": 0.011638942174613476, "reward_std": 0.01774718053638935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0742255225777626, "sampling/sampling_logp_difference/max": 4.726832389831543, "sampling/importance_sampling_ratio/min": 0.008854473941028118, "sampling/importance_sampling_ratio/mean": 0.9967266321182251, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15224116947501898, "clip_ratio/low_mean": 0.03850760939531028, "clip_ratio/low_min": 0.03850760939531028, "clip_ratio/high_mean": 0.011892712675035, "clip_ratio/high_max": 0.011892712675035, "clip_ratio/region_mean": 0.05040032207034528, "reward_total_mean": 0.011638942174613476, "reward_meter_mean": 0.011638942174613476, "reward_meter_std": 0.01774718053638935, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.011638942174613476, "reward_total_composite_std": 0.01774718053638935, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1184.0} {"timestamp_utc": "2026-04-11T21:54:46Z", "mode": "train", "global_step": 1185, "epoch": 0.045759962928637626, "loss": -0.0021, "grad_norm": 0.43357840180397034, "learning_rate": 6.412121212121213e-06, "num_tokens": 2554678.0, "completions/mean_length": 31.0, "completions/min_length": 31.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9985817670822144, "rewards/meter/std": 2.903921813413035e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985817670822144, "rewards/total_composite/std": 2.903921813413035e-05, "reward": 0.9985817670822144, "reward_std": 2.9045202609268017e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004531952552497387, "sampling/sampling_logp_difference/max": 1.0694332122802734, "sampling/importance_sampling_ratio/min": 0.34320297837257385, "sampling/importance_sampling_ratio/mean": 0.9974938631057739, "sampling/importance_sampling_ratio/max": 1.0249570608139038, "entropy": 0.0018207905377494171, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004032257944345474, "reward_total_mean": 0.9985817670822144, "reward_meter_mean": 0.9985817670822144, "reward_meter_std": 2.903921813413035e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985817670822144, "reward_total_composite_std": 2.903921813413035e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1185.0} {"timestamp_utc": "2026-04-11T21:54:52Z", "mode": "train", "global_step": 1186, "epoch": 0.04579857893110905, "loss": -0.0023, "grad_norm": 1.6016870737075806, "learning_rate": 6.40909090909091e-06, "num_tokens": 2556862.0, "completions/mean_length": 98.0, "completions/min_length": 97.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9969850778579712, "rewards/meter/std": 9.93004723568447e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9969850778579712, "rewards/total_composite/std": 9.93004723568447e-05, "reward": 0.9969850778579712, "reward_std": 9.929558291332796e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0031935099977999926, "sampling/sampling_logp_difference/max": 0.7791270017623901, "sampling/importance_sampling_ratio/min": 0.4588063955307007, "sampling/importance_sampling_ratio/mean": 0.9980478882789612, "sampling/importance_sampling_ratio/max": 1.0727391242980957, "entropy": 0.011278128949925303, "clip_ratio/low_mean": 0.0012886597542092204, "clip_ratio/low_min": 0.0012886597542092204, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0012886597542092204, "reward_total_mean": 0.9969850778579712, "reward_meter_mean": 0.9969850778579712, "reward_meter_std": 9.93004723568447e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9969850778579712, "reward_total_composite_std": 9.93004723568447e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1186.0} {"timestamp_utc": "2026-04-11T21:54:57Z", "mode": "train", "global_step": 1187, "epoch": 0.045837194933580475, "loss": 0.0003, "grad_norm": 0.6999150514602661, "learning_rate": 6.406060606060607e-06, "num_tokens": 2559022.0, "completions/mean_length": 109.0, "completions/min_length": 109.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.0, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9984546899795532, "rewards/meter/std": 4.7120189265115187e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984546899795532, "rewards/total_composite/std": 4.7120189265115187e-05, "reward": 0.9984546899795532, "reward_std": 4.7120178351178765e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0013951655710116029, "sampling/sampling_logp_difference/max": 0.15581762790679932, "sampling/importance_sampling_ratio/min": 0.8960135579109192, "sampling/importance_sampling_ratio/mean": 1.000537633895874, "sampling/importance_sampling_ratio/max": 1.168613076210022, "entropy": 0.010229390056338161, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0011467889416962862, "clip_ratio/high_max": 0.0011467889416962862, "clip_ratio/region_mean": 0.0011467889416962862, "reward_total_mean": 0.9984546899795532, "reward_meter_mean": 0.9984546899795532, "reward_meter_std": 4.7120189265115187e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984546899795532, "reward_total_composite_std": 4.7120189265115187e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1187.0} {"timestamp_utc": "2026-04-11T21:55:02Z", "mode": "train", "global_step": 1188, "epoch": 0.0458758109360519, "loss": 0.0782, "grad_norm": 3.00974702835083, "learning_rate": 6.403030303030303e-06, "num_tokens": 2560807.0, "completions/mean_length": 75.125, "completions/min_length": 72.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.125, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.8669778108596802, "rewards/meter/std": 0.342582643032074, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8669778108596802, "rewards/total_composite/std": 0.342582643032074, "reward": 0.8669778108596802, "reward_std": 0.342582643032074, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014270931482315063, "sampling/sampling_logp_difference/max": 3.183518409729004, "sampling/importance_sampling_ratio/min": 0.04143959656357765, "sampling/importance_sampling_ratio/mean": 1.000046730041504, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02158025815151632, "clip_ratio/low_mean": 0.0013736264081671834, "clip_ratio/low_min": 0.0013736264081671834, "clip_ratio/high_mean": 0.006677350495010614, "clip_ratio/high_max": 0.006677350495010614, "clip_ratio/region_mean": 0.008050976903177798, "reward_total_mean": 0.8669778108596802, "reward_meter_mean": 0.8669778108596802, "reward_meter_std": 0.342582643032074, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8669778108596802, "reward_total_composite_std": 0.342582643032074, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1188.0} {"timestamp_utc": "2026-04-11T21:55:07Z", "mode": "train", "global_step": 1189, "epoch": 0.04591442693852332, "loss": 0.0287, "grad_norm": 7.2858405113220215, "learning_rate": 6.4000000000000006e-06, "num_tokens": 2562683.0, "completions/mean_length": 65.5, "completions/min_length": 63.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.4144327640533447, "rewards/meter/std": 0.30921247601509094, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4144327640533447, "rewards/total_composite/std": 0.30921247601509094, "reward": 0.4144327640533447, "reward_std": 0.30921247601509094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035145364701747894, "sampling/sampling_logp_difference/max": 2.764584541320801, "sampling/importance_sampling_ratio/min": 0.06300227344036102, "sampling/importance_sampling_ratio/mean": 1.0029542446136475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10702869668602943, "clip_ratio/low_mean": 0.022366520133800805, "clip_ratio/low_min": 0.022366520133800805, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.024319645133800805, "reward_total_mean": 0.4144327640533447, "reward_meter_mean": 0.4144327640533447, "reward_meter_std": 0.30921247601509094, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4144327640533447, "reward_total_composite_std": 0.30921247601509094, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1189.0} {"timestamp_utc": "2026-04-11T21:55:11Z", "mode": "train", "global_step": 1190, "epoch": 0.04595304294099475, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.396969696969697e-06, "num_tokens": 2564211.0, "completions/mean_length": 31.0, "completions/min_length": 31.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00010159891098737717, "sampling/sampling_logp_difference/max": 0.0022197323851287365, "sampling/importance_sampling_ratio/min": 0.9999852180480957, "sampling/importance_sampling_ratio/mean": 1.0001014471054077, "sampling/importance_sampling_ratio/max": 1.0022221803665161, "entropy": 0.0008445165294688195, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1190.0} {"timestamp_utc": "2026-04-11T21:55:16Z", "mode": "train", "global_step": 1191, "epoch": 0.04599165894346617, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.393939393939394e-06, "num_tokens": 2566107.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 6.144684448372573e-05, "sampling/sampling_logp_difference/max": 0.001986202783882618, "sampling/importance_sampling_ratio/min": 0.999610185623169, "sampling/importance_sampling_ratio/mean": 1.0000598430633545, "sampling/importance_sampling_ratio/max": 1.001988172531128, "entropy": 0.00044536186032928526, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1191.0} {"timestamp_utc": "2026-04-11T21:55:21Z", "mode": "train", "global_step": 1192, "epoch": 0.046030274945937595, "loss": 0.0119, "grad_norm": 4.474715232849121, "learning_rate": 6.390909090909091e-06, "num_tokens": 2568053.0, "completions/mean_length": 64.25, "completions/min_length": 62.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.687461256980896, "rewards/meter/std": 0.23647956550121307, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.687461256980896, "rewards/total_composite/std": 0.23647956550121307, "reward": 0.687461256980896, "reward_std": 0.23647956550121307, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021633058786392212, "sampling/sampling_logp_difference/max": 0.8202657699584961, "sampling/importance_sampling_ratio/min": 0.4403146207332611, "sampling/importance_sampling_ratio/mean": 0.998434841632843, "sampling/importance_sampling_ratio/max": 1.4843707084655762, "entropy": 0.09414011146873236, "clip_ratio/low_mean": 0.005769230774603784, "clip_ratio/low_min": 0.005769230774603784, "clip_ratio/high_mean": 0.021223400719463825, "clip_ratio/high_max": 0.021223400719463825, "clip_ratio/region_mean": 0.02699263149406761, "reward_total_mean": 0.687461256980896, "reward_meter_mean": 0.687461256980896, "reward_meter_std": 0.23647956550121307, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.687461256980896, "reward_total_composite_std": 0.23647956550121307, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1192.0} {"timestamp_utc": "2026-04-11T21:55:26Z", "mode": "train", "global_step": 1193, "epoch": 0.04606889094840902, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.387878787878789e-06, "num_tokens": 2569709.0, "completions/mean_length": 48.0, "completions/min_length": 48.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9887858629226685, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9887858629226685, "rewards/total_composite/std": 0.0, "reward": 0.9887858629226685, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0012214966118335724, "sampling/sampling_logp_difference/max": 0.04612874984741211, "sampling/importance_sampling_ratio/min": 0.9549189805984497, "sampling/importance_sampling_ratio/mean": 1.000608205795288, "sampling/importance_sampling_ratio/max": 1.0443377494812012, "entropy": 0.009338034316897392, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9887858629226685, "reward_meter_mean": 0.9887858629226685, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9887858629226685, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1193.0} {"timestamp_utc": "2026-04-11T21:55:32Z", "mode": "train", "global_step": 1194, "epoch": 0.04610750695088044, "loss": -0.0001, "grad_norm": 0.1992160677909851, "learning_rate": 6.384848484848485e-06, "num_tokens": 2572614.0, "completions/mean_length": 181.125, "completions/min_length": 181.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.125, "completions/min_terminated_length": 181.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.9984710216522217, "rewards/meter/std": 2.833232247212436e-05, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8320591449737549, "rewards/total_composite/std": 2.3610265998286195e-05, "reward": 0.8320591449737549, "reward_std": 2.3623691959073767e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0009484349866397679, "sampling/sampling_logp_difference/max": 0.3491075038909912, "sampling/importance_sampling_ratio/min": 0.7053173184394836, "sampling/importance_sampling_ratio/mean": 0.9998201131820679, "sampling/importance_sampling_ratio/max": 1.209038496017456, "entropy": 0.006256086868233979, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8320591449737549, "reward_meter_mean": 0.9984710216522217, "reward_meter_std": 2.833232247212436e-05, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8320591449737549, "reward_total_composite_std": 2.3610265998286195e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1194.0} {"timestamp_utc": "2026-04-11T21:55:38Z", "mode": "train", "global_step": 1195, "epoch": 0.04614612295335187, "loss": -0.0152, "grad_norm": 2.0341482162475586, "learning_rate": 6.381818181818182e-06, "num_tokens": 2575435.0, "completions/mean_length": 151.625, "completions/min_length": 142.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.625, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.9975845813751221, "rewards/meter/std": 0.0008180320146493614, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7980676889419556, "rewards/total_composite/std": 0.0006544252391904593, "reward": 0.7980676889419556, "reward_std": 0.0006544221541844308, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010841303505003452, "sampling/sampling_logp_difference/max": 0.9098663330078125, "sampling/importance_sampling_ratio/min": 0.40257805585861206, "sampling/importance_sampling_ratio/mean": 1.002147912979126, "sampling/importance_sampling_ratio/max": 1.647913932800293, "entropy": 0.02665393566712737, "clip_ratio/low_mean": 0.007669382495805621, "clip_ratio/low_min": 0.007669382495805621, "clip_ratio/high_mean": 0.004880740132648498, "clip_ratio/high_max": 0.004880740132648498, "clip_ratio/region_mean": 0.012550122628454119, "reward_total_mean": 0.7980676889419556, "reward_meter_mean": 0.9975845813751221, "reward_meter_std": 0.0008180320146493614, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7980676889419556, "reward_total_composite_std": 0.0006544252391904593, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1195.0} {"timestamp_utc": "2026-04-11T21:55:44Z", "mode": "train", "global_step": 1196, "epoch": 0.04618473895582329, "loss": -0.01, "grad_norm": 9.221901893615723, "learning_rate": 6.37878787878788e-06, "num_tokens": 2577632.0, "completions/mean_length": 103.625, "completions/min_length": 100.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.625, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.9974844455718994, "rewards/meter/std": 0.0006991037516854703, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9974844455718994, "rewards/total_composite/std": 0.0006991037516854703, "reward": 0.9974844455718994, "reward_std": 0.0006991035188548267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015732292085886, "sampling/sampling_logp_difference/max": 2.215559244155884, "sampling/importance_sampling_ratio/min": 0.10909249633550644, "sampling/importance_sampling_ratio/mean": 0.9986835718154907, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03731701336801052, "clip_ratio/low_mean": 0.014889341779053211, "clip_ratio/low_min": 0.014889341779053211, "clip_ratio/high_mean": 0.0035601977724581957, "clip_ratio/high_max": 0.0035601977724581957, "clip_ratio/region_mean": 0.018449539551511407, "reward_total_mean": 0.9974844455718994, "reward_meter_mean": 0.9974844455718994, "reward_meter_std": 0.0006991037516854703, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9974844455718994, "reward_total_composite_std": 0.0006991037516854703, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1196.0} {"timestamp_utc": "2026-04-11T21:55:48Z", "mode": "train", "global_step": 1197, "epoch": 0.046223354958294716, "loss": 0.1338, "grad_norm": 24.279144287109375, "learning_rate": 6.375757575757576e-06, "num_tokens": 2579110.0, "completions/mean_length": 35.75, "completions/min_length": 31.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9846537113189697, "rewards/meter/std": 0.032167691737413406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9846537113189697, "rewards/total_composite/std": 0.032167691737413406, "reward": 0.9846537113189697, "reward_std": 0.03216767683625221, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.037709400057792664, "sampling/sampling_logp_difference/max": 3.0687551498413086, "sampling/importance_sampling_ratio/min": 0.04647897556424141, "sampling/importance_sampling_ratio/mean": 0.995938777923584, "sampling/importance_sampling_ratio/max": 1.4753588438034058, "entropy": 0.05178397847339511, "clip_ratio/low_mean": 0.002659574383869767, "clip_ratio/low_min": 0.002659574383869767, "clip_ratio/high_mean": 0.018585751531645656, "clip_ratio/high_max": 0.018585751531645656, "clip_ratio/region_mean": 0.021245325915515423, "reward_total_mean": 0.9846537113189697, "reward_meter_mean": 0.9846537113189697, "reward_meter_std": 0.032167691737413406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9846537113189697, "reward_total_composite_std": 0.032167691737413406, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1197.0} {"timestamp_utc": "2026-04-11T21:55:58Z", "mode": "train", "global_step": 1198, "epoch": 0.04626197096076614, "loss": -0.0433, "grad_norm": 4.205373287200928, "learning_rate": 6.372727272727274e-06, "num_tokens": 2580935.0, "completions/mean_length": 128.125, "completions/min_length": 23.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.28572082519531, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.042788296937942505, "rewards/meter/std": 0.09419418126344681, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.042788296937942505, "rewards/total_composite/std": 0.09419418126344681, "reward": 0.042788296937942505, "reward_std": 0.09419417381286621, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06149071082472801, "sampling/sampling_logp_difference/max": 3.791663408279419, "sampling/importance_sampling_ratio/min": 0.022558048367500305, "sampling/importance_sampling_ratio/mean": 0.9972495436668396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17751457169651985, "clip_ratio/low_mean": 0.025734331109561026, "clip_ratio/low_min": 0.025734331109561026, "clip_ratio/high_mean": 0.001623376621864736, "clip_ratio/high_max": 0.001623376621864736, "clip_ratio/region_mean": 0.027357707731425762, "reward_total_mean": 0.042788296937942505, "reward_meter_mean": 0.042788296937942505, "reward_meter_std": 0.09419418126344681, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.042788296937942505, "reward_total_composite_std": 0.09419418126344681, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1198.0} {"timestamp_utc": "2026-04-11T21:56:04Z", "mode": "train", "global_step": 1199, "epoch": 0.046300586963237564, "loss": 0.0003, "grad_norm": 0.07112119346857071, "learning_rate": 6.3696969696969706e-06, "num_tokens": 2583063.0, "completions/mean_length": 109.0, "completions/min_length": 109.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.0, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9984593391418457, "rewards/meter/std": 2.086405856971396e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984593391418457, "rewards/total_composite/std": 2.086405856971396e-06, "reward": 0.9984593391418457, "reward_std": 2.0953264083800605e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0007174843340180814, "sampling/sampling_logp_difference/max": 0.18243646621704102, "sampling/importance_sampling_ratio/min": 0.8332375884056091, "sampling/importance_sampling_ratio/mean": 1.0001569986343384, "sampling/importance_sampling_ratio/max": 1.0703777074813843, "entropy": 0.006159535580081865, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9984593391418457, "reward_meter_mean": 0.9984593391418457, "reward_meter_std": 2.086405856971396e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984593391418457, "reward_total_composite_std": 2.086405856971396e-06, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1199.0} {"timestamp_utc": "2026-04-11T21:56:08Z", "mode": "train", "global_step": 1200, "epoch": 0.04633920296570899, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.366666666666668e-06, "num_tokens": 2584463.0, "completions/mean_length": 24.0, "completions/min_length": 24.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.0, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.9885647892951965, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9885647892951965, "rewards/total_composite/std": 0.0, "reward": 0.9885647892951965, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006689532892778516, "sampling/sampling_logp_difference/max": 0.01897343248128891, "sampling/importance_sampling_ratio/min": 0.981205403804779, "sampling/importance_sampling_ratio/mean": 1.000182867050171, "sampling/importance_sampling_ratio/max": 1.0128138065338135, "entropy": 0.006478953931946307, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9885647892951965, "reward_meter_mean": 0.9885647892951965, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9885647892951965, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1200.0} {"timestamp_utc": "2026-04-11T21:57:39Z", "mode": "eval", "global_step": 1200, "epoch": 0.04633920296570899, "eval_loss": NaN, "eval_runtime": 90.6551, "eval_samples_per_second": 1.147, "eval_steps_per_second": 0.143, "eval_num_tokens": 2584463.0, "eval_completions/mean_length": 250.20192307692307, "eval_completions/min_length": 58.92307692307692, "eval_completions/max_length": 487.9230769230769, "eval_completions/clipped_ratio": 0.16346153846153846, "eval_completions/mean_terminated_length": 198.65751765324518, "eval_completions/min_terminated_length": 58.92307692307692, "eval_completions/max_terminated_length": 423.6923076923077, "eval_rewards/meter/mean": 0.5284909640367215, "eval_rewards/meter/std": 0.49147624923632693, "eval_rewards/count_adherence/mean": 0.8193528331243075, "eval_rewards/count_adherence/std": 0.28436795794046843, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/total_composite/mean": 0.4919216437981679, "eval_rewards/total_composite/std": 0.4640866976517897, "eval_reward": 0.4919216437981679, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.002731083811690601, "eval_sampling/sampling_logp_difference/max": 0.5404733006770794, "eval_sampling/importance_sampling_ratio/min": 0.6044848309113429, "eval_sampling/importance_sampling_ratio/mean": 1.0002888853733356, "eval_sampling/importance_sampling_ratio/max": 1.286704604442303, "eval_entropy": 0.01765880210754963, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4919216437981679, "eval_reward_meter_mean": 0.5284909640367215, "eval_reward_meter_std": 0.49147624923632693, "eval_reward_count_adherence_mean": 0.8193528331243075, "eval_reward_count_adherence_std": 0.28436795794046843, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_total_composite_mean": 0.4919216437981679, "eval_reward_total_composite_std": 0.4640866976517897, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1200.0} {"timestamp_utc": "2026-04-11T21:57:46Z", "mode": "train", "global_step": 1201, "epoch": 0.04637781896818041, "loss": 0.0132, "grad_norm": 9.135161399841309, "learning_rate": 6.363636363636364e-06, "num_tokens": 2586236.0, "completions/mean_length": 52.625, "completions/min_length": 50.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.6550425291061401, "rewards/meter/std": 0.43107059597969055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6550425291061401, "rewards/total_composite/std": 0.43107059597969055, "reward": 0.6550425291061401, "reward_std": 0.43107059597969055, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03184327110648155, "sampling/sampling_logp_difference/max": 3.4988458156585693, "sampling/importance_sampling_ratio/min": 0.030232258141040802, "sampling/importance_sampling_ratio/mean": 0.9945924282073975, "sampling/importance_sampling_ratio/max": 1.7149361371994019, "entropy": 0.09002233669161797, "clip_ratio/low_mean": 0.007075471803545952, "clip_ratio/low_min": 0.007075471803545952, "clip_ratio/high_mean": 0.024252045433968306, "clip_ratio/high_max": 0.024252045433968306, "clip_ratio/region_mean": 0.03132751723751426, "reward_total_mean": 0.6550425291061401, "reward_meter_mean": 0.6550425291061401, "reward_meter_std": 0.43107059597969055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6550425291061401, "reward_total_composite_std": 0.43107059597969055, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1201.0} {"timestamp_utc": "2026-04-11T21:57:51Z", "mode": "train", "global_step": 1202, "epoch": 0.046416434970651836, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.3606060606060615e-06, "num_tokens": 2588004.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 4.060910941916518e-05, "sampling/sampling_logp_difference/max": 0.0006998751778155565, "sampling/importance_sampling_ratio/min": 0.999674379825592, "sampling/importance_sampling_ratio/mean": 1.0000392198562622, "sampling/importance_sampling_ratio/max": 1.0007001161575317, "entropy": 0.00032512086181668565, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1202.0} {"timestamp_utc": "2026-04-11T21:58:02Z", "mode": "train", "global_step": 1203, "epoch": 0.04645505097312326, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.357575757575758e-06, "num_tokens": 2590148.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9954087138175964, "rewards/meter/std": 0.004101373255252838, "rewards/count_adherence/mean": 0.9333333373069763, "rewards/count_adherence/std": 0.035634830594062805, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9291483163833618, "rewards/total_composite/std": 0.038386568427085876, "reward": 0.9291483163833618, "reward_std": 0.03838656470179558, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9291483163833618, "reward_meter_mean": 0.9954087138175964, "reward_meter_std": 0.004101373255252838, "reward_count_adherence_mean": 0.9333333373069763, "reward_count_adherence_std": 0.035634830594062805, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9291483163833618, "reward_total_composite_std": 0.038386568427085876, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1203.0} {"timestamp_utc": "2026-04-11T21:58:07Z", "mode": "train", "global_step": 1204, "epoch": 0.046493666975594684, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.354545454545455e-06, "num_tokens": 2592036.0, "completions/mean_length": 73.0, "completions/min_length": 73.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.998460054397583, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.998460054397583, "rewards/total_composite/std": 0.0, "reward": 0.998460054397583, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005946997553110123, "sampling/sampling_logp_difference/max": 0.04169168323278427, "sampling/importance_sampling_ratio/min": 0.9929065108299255, "sampling/importance_sampling_ratio/mean": 1.0005292892456055, "sampling/importance_sampling_ratio/max": 1.0425729751586914, "entropy": 0.004701392899733037, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.998460054397583, "reward_meter_mean": 0.998460054397583, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.998460054397583, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1204.0} {"timestamp_utc": "2026-04-11T21:58:13Z", "mode": "train", "global_step": 1205, "epoch": 0.04653228297806611, "loss": 0.0194, "grad_norm": 3.3517215251922607, "learning_rate": 6.3515151515151516e-06, "num_tokens": 2594536.0, "completions/mean_length": 131.5, "completions/min_length": 123.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.5, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9980243444442749, "rewards/meter/std": 0.0006618935731239617, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9980243444442749, "rewards/total_composite/std": 0.0006618935731239617, "reward": 0.9980243444442749, "reward_std": 0.0006618941552005708, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013629158958792686, "sampling/sampling_logp_difference/max": 1.8188239336013794, "sampling/importance_sampling_ratio/min": 0.1622164100408554, "sampling/importance_sampling_ratio/mean": 0.9988694787025452, "sampling/importance_sampling_ratio/max": 1.9299285411834717, "entropy": 0.025803213589824736, "clip_ratio/low_mean": 0.004699248122051358, "clip_ratio/low_min": 0.004699248122051358, "clip_ratio/high_mean": 0.0028809126815758646, "clip_ratio/high_max": 0.0028809126815758646, "clip_ratio/region_mean": 0.007580160803627223, "reward_total_mean": 0.9980243444442749, "reward_meter_mean": 0.9980243444442749, "reward_meter_std": 0.0006618935731239617, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9980243444442749, "reward_total_composite_std": 0.0006618935731239617, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1205.0} {"timestamp_utc": "2026-04-11T21:58:18Z", "mode": "train", "global_step": 1206, "epoch": 0.04657089898053753, "loss": 0.0096, "grad_norm": 0.9980946779251099, "learning_rate": 6.34848484848485e-06, "num_tokens": 2596578.0, "completions/mean_length": 63.25, "completions/min_length": 63.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.25, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9979959726333618, "rewards/meter/std": 0.00013933748414274305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9979959726333618, "rewards/total_composite/std": 0.00013933748414274305, "reward": 0.9979959726333618, "reward_std": 0.00013933748414274305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006114223971962929, "sampling/sampling_logp_difference/max": 0.8598930835723877, "sampling/importance_sampling_ratio/min": 0.8158876299858093, "sampling/importance_sampling_ratio/mean": 1.004797339439392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02251835889182985, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/region_mean": 0.0058302809484303, "reward_total_mean": 0.9979959726333618, "reward_meter_mean": 0.9979959726333618, "reward_meter_std": 0.00013933748414274305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9979959726333618, "reward_total_composite_std": 0.00013933748414274305, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1206.0} {"timestamp_utc": "2026-04-11T21:58:28Z", "mode": "train", "global_step": 1207, "epoch": 0.04660951498300896, "loss": -0.1249, "grad_norm": 2.9109625816345215, "learning_rate": 6.345454545454546e-06, "num_tokens": 2598622.0, "completions/mean_length": 161.5, "completions/min_length": 110.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 111.42857360839844, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.7084651589393616, "rewards/meter/std": 0.43425631523132324, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.35634833574295044, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.708056628704071, "rewards/total_composite/std": 0.43500834703445435, "reward": 0.708056628704071, "reward_std": 0.43500831723213196, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01707235537469387, "sampling/sampling_logp_difference/max": 3.4127614498138428, "sampling/importance_sampling_ratio/min": 0.032950084656476974, "sampling/importance_sampling_ratio/mean": 0.996599018573761, "sampling/importance_sampling_ratio/max": 1.323562502861023, "entropy": 0.03559571597725153, "clip_ratio/low_mean": 0.0062500000931322575, "clip_ratio/low_min": 0.0062500000931322575, "clip_ratio/high_mean": 0.004545454401522875, "clip_ratio/high_max": 0.004545454401522875, "clip_ratio/region_mean": 0.010795454494655132, "reward_total_mean": 0.708056628704071, "reward_meter_mean": 0.7084651589393616, "reward_meter_std": 0.43425631523132324, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.35634833574295044, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.708056628704071, "reward_total_composite_std": 0.43500834703445435, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1207.0} {"timestamp_utc": "2026-04-11T21:58:33Z", "mode": "train", "global_step": 1208, "epoch": 0.04664813098548038, "loss": -0.0475, "grad_norm": 3.5151166915893555, "learning_rate": 6.342424242424243e-06, "num_tokens": 2601064.0, "completions/mean_length": 139.25, "completions/min_length": 127.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.25, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9982051849365234, "rewards/meter/std": 0.0006945931818336248, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9046633243560791, "rewards/total_composite/std": 0.12945260107517242, "reward": 0.9046633243560791, "reward_std": 0.12945258617401123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005466024857014418, "sampling/sampling_logp_difference/max": 1.0940611362457275, "sampling/importance_sampling_ratio/min": 0.3909231126308441, "sampling/importance_sampling_ratio/mean": 1.001442790031433, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.014410868810955435, "clip_ratio/low_mean": 0.0019170877640135586, "clip_ratio/low_min": 0.0019170877640135586, "clip_ratio/high_mean": 0.0017241379246115685, "clip_ratio/high_max": 0.0017241379246115685, "clip_ratio/region_mean": 0.003641225688625127, "reward_total_mean": 0.9046633243560791, "reward_meter_mean": 0.9982051849365234, "reward_meter_std": 0.0006945931818336248, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9046633243560791, "reward_total_composite_std": 0.12945260107517242, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1208.0} {"timestamp_utc": "2026-04-11T21:58:38Z", "mode": "train", "global_step": 1209, "epoch": 0.046686746987951805, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.33939393939394e-06, "num_tokens": 2602944.0, "completions/mean_length": 73.0, "completions/min_length": 73.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.998460054397583, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.998460054397583, "rewards/total_composite/std": 0.0, "reward": 0.998460054397583, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0001915806788019836, "sampling/sampling_logp_difference/max": 0.006884189322590828, "sampling/importance_sampling_ratio/min": 0.9943536520004272, "sampling/importance_sampling_ratio/mean": 1.000165343284607, "sampling/importance_sampling_ratio/max": 1.0069079399108887, "entropy": 0.002016735728830099, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.998460054397583, "reward_meter_mean": 0.998460054397583, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.998460054397583, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1209.0} {"timestamp_utc": "2026-04-11T21:58:46Z", "mode": "train", "global_step": 1210, "epoch": 0.04672536299042323, "loss": 0.0672, "grad_norm": 3.3016936779022217, "learning_rate": 6.336363636363637e-06, "num_tokens": 2606314.0, "completions/mean_length": 237.25, "completions/min_length": 213.0, "completions/max_length": 305.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 237.25, "completions/min_terminated_length": 213.0, "completions/max_terminated_length": 305.0, "rewards/meter/mean": 0.5468891859054565, "rewards/meter/std": 0.46307867765426636, "rewards/count_adherence/mean": 0.890625, "rewards/count_adherence/std": 0.12387890368700027, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5222339630126953, "rewards/total_composite/std": 0.4492309093475342, "reward": 0.5222339630126953, "reward_std": 0.4492309093475342, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01383388414978981, "sampling/sampling_logp_difference/max": 3.6984915733337402, "sampling/importance_sampling_ratio/min": 0.02476084791123867, "sampling/importance_sampling_ratio/mean": 0.9978852868080139, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03227660758420825, "clip_ratio/low_mean": 0.0032487863209098577, "clip_ratio/low_min": 0.0032487863209098577, "clip_ratio/high_mean": 0.006558712921105325, "clip_ratio/high_max": 0.006558712921105325, "clip_ratio/region_mean": 0.009807499242015183, "reward_total_mean": 0.5222339630126953, "reward_meter_mean": 0.5468891859054565, "reward_meter_std": 0.46307867765426636, "reward_count_adherence_mean": 0.890625, "reward_count_adherence_std": 0.12387890368700027, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5222339630126953, "reward_total_composite_std": 0.4492309093475342, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1210.0} {"timestamp_utc": "2026-04-11T21:58:51Z", "mode": "train", "global_step": 1211, "epoch": 0.04676397899289465, "loss": 0.0248, "grad_norm": 6.198331832885742, "learning_rate": 6.333333333333333e-06, "num_tokens": 2608244.0, "completions/mean_length": 74.25, "completions/min_length": 68.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.6525086164474487, "rewards/meter/std": 0.4410209357738495, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6525086164474487, "rewards/total_composite/std": 0.4410209357738495, "reward": 0.6525086164474487, "reward_std": 0.4410209357738495, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032058991491794586, "sampling/sampling_logp_difference/max": 1.5635433197021484, "sampling/importance_sampling_ratio/min": 0.20939281582832336, "sampling/importance_sampling_ratio/mean": 1.008939504623413, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06842577271163464, "clip_ratio/low_mean": 0.013784109032712877, "clip_ratio/low_min": 0.013784109032712877, "clip_ratio/high_mean": 0.010183869511820376, "clip_ratio/high_max": 0.010183869511820376, "clip_ratio/region_mean": 0.023967978544533253, "reward_total_mean": 0.6525086164474487, "reward_meter_mean": 0.6525086164474487, "reward_meter_std": 0.4410209357738495, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6525086164474487, "reward_total_composite_std": 0.4410209357738495, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1211.0} {"timestamp_utc": "2026-04-11T21:58:56Z", "mode": "train", "global_step": 1212, "epoch": 0.04680259499536608, "loss": -0.0403, "grad_norm": 4.241539001464844, "learning_rate": 6.330303030303031e-06, "num_tokens": 2610066.0, "completions/mean_length": 71.75, "completions/min_length": 64.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.802007794380188, "rewards/meter/std": 0.3227885365486145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.802007794380188, "rewards/total_composite/std": 0.3227885365486145, "reward": 0.802007794380188, "reward_std": 0.3227885067462921, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01067862194031477, "sampling/sampling_logp_difference/max": 0.8669040203094482, "sampling/importance_sampling_ratio/min": 0.42025062441825867, "sampling/importance_sampling_ratio/mean": 1.001684308052063, "sampling/importance_sampling_ratio/max": 1.8988091945648193, "entropy": 0.03452354692853987, "clip_ratio/low_mean": 0.005859375, "clip_ratio/low_min": 0.005859375, "clip_ratio/high_mean": 0.012113899807445705, "clip_ratio/high_max": 0.012113899807445705, "clip_ratio/region_mean": 0.017973274807445705, "reward_total_mean": 0.802007794380188, "reward_meter_mean": 0.802007794380188, "reward_meter_std": 0.3227885365486145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.802007794380188, "reward_total_composite_std": 0.3227885365486145, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1212.0} {"timestamp_utc": "2026-04-11T21:59:02Z", "mode": "train", "global_step": 1213, "epoch": 0.0468412109978375, "loss": 0.0118, "grad_norm": 2.2606401443481445, "learning_rate": 6.327272727272727e-06, "num_tokens": 2613099.0, "completions/mean_length": 178.125, "completions/min_length": 177.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 178.125, "completions/min_terminated_length": 177.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.3571264147758484, "rewards/meter/std": 0.16085447371006012, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3004804849624634, "rewards/total_composite/std": 0.15694603323936462, "reward": 0.3004804849624634, "reward_std": 0.15694601833820343, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03583236038684845, "sampling/sampling_logp_difference/max": 8.81896686553955, "sampling/importance_sampling_ratio/min": 0.00014790108252782375, "sampling/importance_sampling_ratio/mean": 0.9948055148124695, "sampling/importance_sampling_ratio/max": 1.5341099500656128, "entropy": 0.0576698393560946, "clip_ratio/low_mean": 0.006967985071241856, "clip_ratio/low_min": 0.006967985071241856, "clip_ratio/high_mean": 0.0063479969976469874, "clip_ratio/high_max": 0.0063479969976469874, "clip_ratio/region_mean": 0.013315982068888843, "reward_total_mean": 0.3004804849624634, "reward_meter_mean": 0.3571264147758484, "reward_meter_std": 0.16085447371006012, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3004804849624634, "reward_total_composite_std": 0.15694603323936462, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1213.0} {"timestamp_utc": "2026-04-11T21:59:12Z", "mode": "train", "global_step": 1214, "epoch": 0.046879827000308925, "loss": -0.1967, "grad_norm": 0.24958960711956024, "learning_rate": 6.324242424242425e-06, "num_tokens": 2615234.0, "completions/mean_length": 145.875, "completions/min_length": 93.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 93.5714340209961, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9934712052345276, "rewards/meter/std": 0.011593195609748363, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9934712052345276, "rewards/total_composite/std": 0.011593195609748363, "reward": 0.9934712052345276, "reward_std": 0.011593214236199856, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00628403015434742, "sampling/sampling_logp_difference/max": 1.1038224697113037, "sampling/importance_sampling_ratio/min": 0.3316011130809784, "sampling/importance_sampling_ratio/mean": 1.0019924640655518, "sampling/importance_sampling_ratio/max": 1.9223887920379639, "entropy": 0.012194187263958156, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.002659574383869767, "clip_ratio/high_max": 0.002659574383869767, "clip_ratio/region_mean": 0.002659574383869767, "reward_total_mean": 0.9934712052345276, "reward_meter_mean": 0.9934712052345276, "reward_meter_std": 0.011593195609748363, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9934712052345276, "reward_total_composite_std": 0.011593195609748363, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1214.0} {"timestamp_utc": "2026-04-11T21:59:17Z", "mode": "train", "global_step": 1215, "epoch": 0.04691844300278035, "loss": 0.0017, "grad_norm": 0.5396441221237183, "learning_rate": 6.3212121212121216e-06, "num_tokens": 2617332.0, "completions/mean_length": 93.25, "completions/min_length": 93.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.25, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9976124167442322, "rewards/meter/std": 5.835621414007619e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9976124167442322, "rewards/total_composite/std": 5.835621414007619e-05, "reward": 0.9976124167442322, "reward_std": 5.836541095050052e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004978191573172808, "sampling/sampling_logp_difference/max": 1.4988858699798584, "sampling/importance_sampling_ratio/min": 0.22337891161441803, "sampling/importance_sampling_ratio/mean": 1.0000784397125244, "sampling/importance_sampling_ratio/max": 1.5586631298065186, "entropy": 0.013329520239494741, "clip_ratio/low_mean": 0.002659574383869767, "clip_ratio/low_min": 0.002659574383869767, "clip_ratio/high_mean": 0.0013440860202535987, "clip_ratio/high_max": 0.0013440860202535987, "clip_ratio/region_mean": 0.004003660404123366, "reward_total_mean": 0.9976124167442322, "reward_meter_mean": 0.9976124167442322, "reward_meter_std": 5.835621414007619e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9976124167442322, "reward_total_composite_std": 5.835621414007619e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1215.0} {"timestamp_utc": "2026-04-11T21:59:22Z", "mode": "train", "global_step": 1216, "epoch": 0.046957059005251774, "loss": 0.0011, "grad_norm": 1.1965175867080688, "learning_rate": 6.318181818181819e-06, "num_tokens": 2619204.0, "completions/mean_length": 72.0, "completions/min_length": 72.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9889691472053528, "rewards/meter/std": 0.00030986362253315747, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9889691472053528, "rewards/total_composite/std": 0.00030986362253315747, "reward": 0.9889691472053528, "reward_std": 0.00030986362253315747, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011142927221953869, "sampling/sampling_logp_difference/max": 2.433090925216675, "sampling/importance_sampling_ratio/min": 0.08776513487100601, "sampling/importance_sampling_ratio/mean": 0.9968021512031555, "sampling/importance_sampling_ratio/max": 1.3431813716888428, "entropy": 0.018052175058983266, "clip_ratio/low_mean": 0.010416666744276881, "clip_ratio/low_min": 0.010416666744276881, "clip_ratio/high_mean": 0.0034722222480922937, "clip_ratio/high_max": 0.0034722222480922937, "clip_ratio/region_mean": 0.013888888992369175, "reward_total_mean": 0.9889691472053528, "reward_meter_mean": 0.9889691472053528, "reward_meter_std": 0.00030986362253315747, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9889691472053528, "reward_total_composite_std": 0.00030986362253315747, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1216.0} {"timestamp_utc": "2026-04-11T21:59:27Z", "mode": "train", "global_step": 1217, "epoch": 0.0469956750077232, "loss": 0.0982, "grad_norm": 8.206745147705078, "learning_rate": 6.315151515151515e-06, "num_tokens": 2620888.0, "completions/mean_length": 52.5, "completions/min_length": 46.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.5, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.13825760781764984, "rewards/meter/std": 0.18574245274066925, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.13825760781764984, "rewards/total_composite/std": 0.18574245274066925, "reward": 0.13825760781764984, "reward_std": 0.18574243783950806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039606355130672455, "sampling/sampling_logp_difference/max": 1.4499194622039795, "sampling/importance_sampling_ratio/min": 0.23458918929100037, "sampling/importance_sampling_ratio/mean": 0.9966353178024292, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10656908107921481, "clip_ratio/low_mean": 0.01640390558168292, "clip_ratio/low_min": 0.01640390558168292, "clip_ratio/high_mean": 0.007925724843516946, "clip_ratio/high_max": 0.007925724843516946, "clip_ratio/region_mean": 0.024329630425199866, "reward_total_mean": 0.13825760781764984, "reward_meter_mean": 0.13825760781764984, "reward_meter_std": 0.18574245274066925, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.13825760781764984, "reward_total_composite_std": 0.18574245274066925, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1217.0} {"timestamp_utc": "2026-04-11T21:59:32Z", "mode": "train", "global_step": 1218, "epoch": 0.04703429101019462, "loss": 0.1259, "grad_norm": 9.658549308776855, "learning_rate": 6.3121212121212125e-06, "num_tokens": 2622621.0, "completions/mean_length": 59.625, "completions/min_length": 54.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.625, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.7469707131385803, "rewards/meter/std": 0.1834370195865631, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7469707131385803, "rewards/total_composite/std": 0.1834370195865631, "reward": 0.7469707131385803, "reward_std": 0.1834370195865631, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041233424097299576, "sampling/sampling_logp_difference/max": 2.798147678375244, "sampling/importance_sampling_ratio/min": 0.06092280521988869, "sampling/importance_sampling_ratio/mean": 0.987882137298584, "sampling/importance_sampling_ratio/max": 1.8506815433502197, "entropy": 0.08589892974123359, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.022627133643254638, "clip_ratio/high_max": 0.022627133643254638, "clip_ratio/region_mean": 0.026005512103438377, "reward_total_mean": 0.7469707131385803, "reward_meter_mean": 0.7469707131385803, "reward_meter_std": 0.1834370195865631, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7469707131385803, "reward_total_composite_std": 0.1834370195865631, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1218.0} {"timestamp_utc": "2026-04-11T21:59:41Z", "mode": "train", "global_step": 1219, "epoch": 0.047072907012666046, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.309090909090909e-06, "num_tokens": 2628104.0, "completions/mean_length": 432.375, "completions/min_length": 423.0, "completions/max_length": 438.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 432.375, "completions/min_terminated_length": 423.0, "completions/max_terminated_length": 438.0, "rewards/meter/mean": 0.998979389667511, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.7272727489471436, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7265304327011108, "rewards/total_composite/std": 0.0, "reward": 0.7265304327011108, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0029552343767136335, "sampling/sampling_logp_difference/max": 1.7289729118347168, "sampling/importance_sampling_ratio/min": 0.17746658623218536, "sampling/importance_sampling_ratio/mean": 1.0007617473602295, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.008362318098079413, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7265304327011108, "reward_meter_mean": 0.998979389667511, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.7272727489471436, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7265304327011108, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1219.0} {"timestamp_utc": "2026-04-11T21:59:47Z", "mode": "train", "global_step": 1220, "epoch": 0.04711152301513747, "loss": 0.0438, "grad_norm": 8.604512214660645, "learning_rate": 6.306060606060607e-06, "num_tokens": 2630484.0, "completions/mean_length": 147.5, "completions/min_length": 120.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.5, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.47342073917388916, "rewards/meter/std": 0.11361439526081085, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.15430334210395813, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3546171188354492, "rewards/total_composite/std": 0.10311604291200638, "reward": 0.3546171188354492, "reward_std": 0.10311603546142578, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02061653509736061, "sampling/sampling_logp_difference/max": 2.449690103530884, "sampling/importance_sampling_ratio/min": 0.08632033318281174, "sampling/importance_sampling_ratio/mean": 1.0022088289260864, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07398146251216531, "clip_ratio/low_mean": 0.006893568090163171, "clip_ratio/low_min": 0.006893568090163171, "clip_ratio/high_mean": 0.0015822785208001733, "clip_ratio/high_max": 0.0015822785208001733, "clip_ratio/region_mean": 0.008475846610963345, "reward_total_mean": 0.3546171188354492, "reward_meter_mean": 0.47342073917388916, "reward_meter_std": 0.11361439526081085, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.15430334210395813, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3546171188354492, "reward_total_composite_std": 0.10311604291200638, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1220.0} {"timestamp_utc": "2026-04-11T21:59:57Z", "mode": "train", "global_step": 1221, "epoch": 0.047150139017608894, "loss": -0.1113, "grad_norm": 0.906308650970459, "learning_rate": 6.303030303030303e-06, "num_tokens": 2631995.0, "completions/mean_length": 94.875, "completions/min_length": 32.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 35.28571701049805, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9913299083709717, "rewards/meter/std": 0.020195109769701958, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.8736585378646851, "rewards/total_composite/std": 0.3530118465423584, "reward": 0.8736585378646851, "reward_std": 0.353011816740036, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06524569541215897, "sampling/sampling_logp_difference/max": 6.447518825531006, "sampling/importance_sampling_ratio/min": 0.0015844486188143492, "sampling/importance_sampling_ratio/mean": 0.9859198927879333, "sampling/importance_sampling_ratio/max": 1.9141241312026978, "entropy": 0.08671777741983533, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.03222071169875562, "clip_ratio/high_max": 0.03222071169875562, "clip_ratio/region_mean": 0.03222071169875562, "reward_total_mean": 0.8736585378646851, "reward_meter_mean": 0.9913299083709717, "reward_meter_std": 0.020195109769701958, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.8736585378646851, "reward_total_composite_std": 0.3530118465423584, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1221.0} {"timestamp_utc": "2026-04-11T22:00:02Z", "mode": "train", "global_step": 1222, "epoch": 0.04718875502008032, "loss": -0.015, "grad_norm": 5.500716209411621, "learning_rate": 6.300000000000001e-06, "num_tokens": 2633684.0, "completions/mean_length": 56.125, "completions/min_length": 51.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8193844556808472, "rewards/meter/std": 0.22898495197296143, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8193844556808472, "rewards/total_composite/std": 0.22898495197296143, "reward": 0.8193844556808472, "reward_std": 0.22898495197296143, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.049046412110328674, "sampling/sampling_logp_difference/max": 3.637985944747925, "sampling/importance_sampling_ratio/min": 0.026305271312594414, "sampling/importance_sampling_ratio/mean": 0.9902265667915344, "sampling/importance_sampling_ratio/max": 1.5700050592422485, "entropy": 0.0667776691261679, "clip_ratio/low_mean": 0.0024509804788976908, "clip_ratio/low_min": 0.0024509804788976908, "clip_ratio/high_mean": 0.02054290263913572, "clip_ratio/high_max": 0.02054290263913572, "clip_ratio/region_mean": 0.02299388311803341, "reward_total_mean": 0.8193844556808472, "reward_meter_mean": 0.8193844556808472, "reward_meter_std": 0.22898495197296143, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8193844556808472, "reward_total_composite_std": 0.22898495197296143, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1222.0} {"timestamp_utc": "2026-04-11T22:00:07Z", "mode": "train", "global_step": 1223, "epoch": 0.04722737102255174, "loss": 0.0025, "grad_norm": 7.78162956237793, "learning_rate": 6.296969696969697e-06, "num_tokens": 2635388.0, "completions/mean_length": 51.0, "completions/min_length": 51.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8671376705169678, "rewards/meter/std": 0.020614558830857277, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8671376705169678, "rewards/total_composite/std": 0.020614558830857277, "reward": 0.8671376705169678, "reward_std": 0.02061455138027668, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01923941634595394, "sampling/sampling_logp_difference/max": 1.7128831148147583, "sampling/importance_sampling_ratio/min": 0.1803450882434845, "sampling/importance_sampling_ratio/mean": 0.9968818426132202, "sampling/importance_sampling_ratio/max": 1.4597183465957642, "entropy": 0.042778775095939636, "clip_ratio/low_mean": 0.0049019609577953815, "clip_ratio/low_min": 0.0049019609577953815, "clip_ratio/high_mean": 0.014705882640555501, "clip_ratio/high_max": 0.014705882640555501, "clip_ratio/region_mean": 0.019607843598350883, "reward_total_mean": 0.8671376705169678, "reward_meter_mean": 0.8671376705169678, "reward_meter_std": 0.020614558830857277, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8671376705169678, "reward_total_composite_std": 0.020614558830857277, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1223.0} {"timestamp_utc": "2026-04-11T22:00:12Z", "mode": "train", "global_step": 1224, "epoch": 0.047265987025023166, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.293939393939394e-06, "num_tokens": 2637236.0, "completions/mean_length": 73.0, "completions/min_length": 73.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.998460054397583, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.998460054397583, "rewards/total_composite/std": 0.0, "reward": 0.998460054397583, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0007475087768398225, "sampling/sampling_logp_difference/max": 0.09313339740037918, "sampling/importance_sampling_ratio/min": 0.9110719561576843, "sampling/importance_sampling_ratio/mean": 0.9994363784790039, "sampling/importance_sampling_ratio/max": 1.0075687170028687, "entropy": 0.003402701644517947, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.998460054397583, "reward_meter_mean": 0.998460054397583, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.998460054397583, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1224.0} {"timestamp_utc": "2026-04-11T22:00:22Z", "mode": "train", "global_step": 1225, "epoch": 0.04730460302749459, "loss": -0.0807, "grad_norm": 2.4841148853302, "learning_rate": 6.290909090909092e-06, "num_tokens": 2638974.0, "completions/mean_length": 126.25, "completions/min_length": 64.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 71.14286041259766, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.22189268469810486, "rewards/meter/std": 0.1687399446964264, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.22180022299289703, "rewards/total_composite/std": 0.16887813806533813, "reward": 0.22180022299289703, "reward_std": 0.16887813806533813, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03207707777619362, "sampling/sampling_logp_difference/max": 2.4870667457580566, "sampling/importance_sampling_ratio/min": 0.08315351605415344, "sampling/importance_sampling_ratio/mean": 0.9992117881774902, "sampling/importance_sampling_ratio/max": 1.6487122774124146, "entropy": 0.1300158714875579, "clip_ratio/low_mean": 0.0053571430034935474, "clip_ratio/low_min": 0.0053571430034935474, "clip_ratio/high_mean": 0.01835244067478925, "clip_ratio/high_max": 0.01835244067478925, "clip_ratio/region_mean": 0.023709583678282797, "reward_total_mean": 0.22180022299289703, "reward_meter_mean": 0.22189268469810486, "reward_meter_std": 0.1687399446964264, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.22180022299289703, "reward_total_composite_std": 0.16887813806533813, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1225.0} {"timestamp_utc": "2026-04-11T22:00:26Z", "mode": "train", "global_step": 1226, "epoch": 0.047343219029966015, "loss": 0.0017, "grad_norm": 6.133970737457275, "learning_rate": 6.287878787878788e-06, "num_tokens": 2640426.0, "completions/mean_length": 30.5, "completions/min_length": 30.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.996648907661438, "rewards/meter/std": 0.0002604323090054095, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.996648907661438, "rewards/total_composite/std": 0.0002604323090054095, "reward": 0.996648907661438, "reward_std": 0.00026042849640361965, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039376165717840195, "sampling/sampling_logp_difference/max": 1.0917110443115234, "sampling/importance_sampling_ratio/min": 0.3356417119503021, "sampling/importance_sampling_ratio/mean": 0.9833301901817322, "sampling/importance_sampling_ratio/max": 1.7827787399291992, "entropy": 0.07284456863999367, "clip_ratio/low_mean": 0.016397849656641483, "clip_ratio/low_min": 0.016397849656641483, "clip_ratio/high_mean": 0.01587701588869095, "clip_ratio/high_max": 0.01587701588869095, "clip_ratio/region_mean": 0.03227486554533243, "reward_total_mean": 0.996648907661438, "reward_meter_mean": 0.996648907661438, "reward_meter_std": 0.0002604323090054095, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.996648907661438, "reward_total_composite_std": 0.0002604323090054095, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1226.0} {"timestamp_utc": "2026-04-11T22:00:32Z", "mode": "train", "global_step": 1227, "epoch": 0.04738183503243744, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.284848484848486e-06, "num_tokens": 2643410.0, "completions/mean_length": 196.0, "completions/min_length": 196.0, "completions/max_length": 196.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 196.0, "completions/min_terminated_length": 196.0, "completions/max_terminated_length": 196.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7988736629486084, "rewards/total_composite/std": 0.0, "reward": 0.7988736629486084, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.004949238151311874, "sampling/sampling_logp_difference/max": 5.144115447998047, "sampling/importance_sampling_ratio/min": 0.005833631847053766, "sampling/importance_sampling_ratio/mean": 0.9987924098968506, "sampling/importance_sampling_ratio/max": 1.0011606216430664, "entropy": 0.0006115238029451575, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7988736629486084, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7988736629486084, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1227.0} {"timestamp_utc": "2026-04-11T22:00:38Z", "mode": "train", "global_step": 1228, "epoch": 0.04742045103490886, "loss": -0.0037, "grad_norm": 3.007474184036255, "learning_rate": 6.2818181818181825e-06, "num_tokens": 2645046.0, "completions/mean_length": 64.5, "completions/min_length": 63.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9988840222358704, "rewards/meter/std": 0.0006037719431333244, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9988840222358704, "rewards/total_composite/std": 0.0006037719431333244, "reward": 0.9988840222358704, "reward_std": 0.0006037719431333244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02234082669019699, "sampling/sampling_logp_difference/max": 2.8550028800964355, "sampling/importance_sampling_ratio/min": 0.057555653154850006, "sampling/importance_sampling_ratio/mean": 0.9949737191200256, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05550367524847388, "clip_ratio/low_mean": 0.0059523810632526875, "clip_ratio/low_min": 0.0059523810632526875, "clip_ratio/high_mean": 0.011214631143957376, "clip_ratio/high_max": 0.011214631143957376, "clip_ratio/region_mean": 0.017167012207210064, "reward_total_mean": 0.9988840222358704, "reward_meter_mean": 0.9988840222358704, "reward_meter_std": 0.0006037719431333244, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9988840222358704, "reward_total_composite_std": 0.0006037719431333244, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1228.0} {"timestamp_utc": "2026-04-11T22:00:49Z", "mode": "train", "global_step": 1229, "epoch": 0.04745906703738029, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.27878787878788e-06, "num_tokens": 2647006.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9988684058189392, "rewards/meter/std": 0.00021612727141473442, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0391591340303421, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8740172386169434, "rewards/total_composite/std": 0.03929149731993675, "reward": 0.8740172386169434, "reward_std": 0.03929150104522705, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8740172386169434, "reward_meter_mean": 0.9988684058189392, "reward_meter_std": 0.00021612727141473442, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0391591340303421, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8740172386169434, "reward_total_composite_std": 0.03929149731993675, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1229.0} {"timestamp_utc": "2026-04-11T22:00:59Z", "mode": "train", "global_step": 1230, "epoch": 0.04749768303985171, "loss": -0.1469, "grad_norm": 1.649530291557312, "learning_rate": 6.275757575757576e-06, "num_tokens": 2649330.0, "completions/mean_length": 184.5, "completions/min_length": 128.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 137.71429443359375, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.3385840654373169, "rewards/meter/std": 0.24243442714214325, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.30860671401023865, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.2500231862068176, "rewards/total_composite/std": 0.16611215472221375, "reward": 0.2500231862068176, "reward_std": 0.16611215472221375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019756808876991272, "sampling/sampling_logp_difference/max": 4.895002841949463, "sampling/importance_sampling_ratio/min": 0.007483887951821089, "sampling/importance_sampling_ratio/mean": 1.0006352663040161, "sampling/importance_sampling_ratio/max": 1.8365625143051147, "entropy": 0.05436281254515052, "clip_ratio/low_mean": 0.0054711957927793264, "clip_ratio/low_min": 0.0054711957927793264, "clip_ratio/high_mean": 0.005169942509382963, "clip_ratio/high_max": 0.005169942509382963, "clip_ratio/region_mean": 0.01064113830216229, "reward_total_mean": 0.2500231862068176, "reward_meter_mean": 0.3385840654373169, "reward_meter_std": 0.24243442714214325, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.30860671401023865, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.2500231862068176, "reward_total_composite_std": 0.16611215472221375, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1230.0} {"timestamp_utc": "2026-04-11T22:01:04Z", "mode": "train", "global_step": 1231, "epoch": 0.04753629904232314, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.2727272727272734e-06, "num_tokens": 2651706.0, "completions/mean_length": 121.0, "completions/min_length": 121.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.0, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6657280325889587, "rewards/total_composite/std": 0.0, "reward": 0.6657280325889587, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 3.28192618326284e-05, "sampling/sampling_logp_difference/max": 0.0007492472068406641, "sampling/importance_sampling_ratio/min": 0.9993756413459778, "sampling/importance_sampling_ratio/mean": 1.0000293254852295, "sampling/importance_sampling_ratio/max": 1.0007495880126953, "entropy": 0.0002726803577388637, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.6657280325889587, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6657280325889587, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1231.0} {"timestamp_utc": "2026-04-11T22:01:09Z", "mode": "train", "global_step": 1232, "epoch": 0.047574915044794566, "loss": 0.0619, "grad_norm": 13.347013473510742, "learning_rate": 6.26969696969697e-06, "num_tokens": 2653503.0, "completions/mean_length": 75.625, "completions/min_length": 65.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.625, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.14627057313919067, "rewards/meter/std": 0.12811000645160675, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.14627057313919067, "rewards/total_composite/std": 0.12811000645160675, "reward": 0.14627057313919067, "reward_std": 0.12811000645160675, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025237806141376495, "sampling/sampling_logp_difference/max": 1.1749186515808105, "sampling/importance_sampling_ratio/min": 0.30884408950805664, "sampling/importance_sampling_ratio/mean": 1.0060153007507324, "sampling/importance_sampling_ratio/max": 1.5295863151550293, "entropy": 0.10356379672884941, "clip_ratio/low_mean": 0.01266330131329596, "clip_ratio/low_min": 0.01266330131329596, "clip_ratio/high_mean": 0.005769230774603784, "clip_ratio/high_max": 0.005769230774603784, "clip_ratio/region_mean": 0.018432532087899745, "reward_total_mean": 0.14627057313919067, "reward_meter_mean": 0.14627057313919067, "reward_meter_std": 0.12811000645160675, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.14627057313919067, "reward_total_composite_std": 0.12811000645160675, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1232.0} {"timestamp_utc": "2026-04-11T22:01:16Z", "mode": "train", "global_step": 1233, "epoch": 0.04761353104726599, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.266666666666668e-06, "num_tokens": 2657311.0, "completions/mean_length": 289.0, "completions/min_length": 289.0, "completions/max_length": 289.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 289.0, "completions/min_terminated_length": 289.0, "completions/max_terminated_length": 289.0, "rewards/meter/mean": 0.998460054397583, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8558229207992554, "rewards/total_composite/std": 0.0, "reward": 0.8558229207992554, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 8.391581650357693e-05, "sampling/sampling_logp_difference/max": 0.023856770247220993, "sampling/importance_sampling_ratio/min": 0.9764255285263062, "sampling/importance_sampling_ratio/mean": 1.000006914138794, "sampling/importance_sampling_ratio/max": 1.0143555402755737, "entropy": 0.0006672077724942937, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8558229207992554, "reward_meter_mean": 0.998460054397583, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8558229207992554, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1233.0} {"timestamp_utc": "2026-04-11T22:01:21Z", "mode": "train", "global_step": 1234, "epoch": 0.047652147049737414, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.263636363636364e-06, "num_tokens": 2658831.0, "completions/mean_length": 31.0, "completions/min_length": 31.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.030348138883709908, "sampling/sampling_logp_difference/max": 5.0875701904296875, "sampling/importance_sampling_ratio/min": 0.0061730011366307735, "sampling/importance_sampling_ratio/mean": 0.9925329685211182, "sampling/importance_sampling_ratio/max": 1.0111511945724487, "entropy": 0.008347092385520227, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1234.0} {"timestamp_utc": "2026-04-11T22:01:26Z", "mode": "train", "global_step": 1235, "epoch": 0.04769076305220884, "loss": 0.0125, "grad_norm": 4.2324299812316895, "learning_rate": 6.260606060606062e-06, "num_tokens": 2660871.0, "completions/mean_length": 86.0, "completions/min_length": 84.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.0, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.22920642793178558, "rewards/meter/std": 0.040649767965078354, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.15280428528785706, "rewards/total_composite/std": 0.027099840342998505, "reward": 0.15280428528785706, "reward_std": 0.027099842205643654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011969751678407192, "sampling/sampling_logp_difference/max": 1.3204421997070312, "sampling/importance_sampling_ratio/min": 0.2670172154903412, "sampling/importance_sampling_ratio/mean": 1.0045161247253418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0405436703003943, "clip_ratio/low_mean": 0.005797246703878045, "clip_ratio/low_min": 0.005797246703878045, "clip_ratio/high_mean": 0.002941583632491529, "clip_ratio/high_max": 0.002941583632491529, "clip_ratio/region_mean": 0.008738830336369574, "reward_total_mean": 0.15280428528785706, "reward_meter_mean": 0.22920642793178558, "reward_meter_std": 0.040649767965078354, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.15280428528785706, "reward_total_composite_std": 0.027099840342998505, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1235.0} {"timestamp_utc": "2026-04-11T22:01:31Z", "mode": "train", "global_step": 1236, "epoch": 0.04772937905468026, "loss": 0.0426, "grad_norm": 6.601499080657959, "learning_rate": 6.257575757575758e-06, "num_tokens": 2662367.0, "completions/mean_length": 37.0, "completions/min_length": 34.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.20092102885246277, "rewards/meter/std": 0.348273903131485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.20092102885246277, "rewards/total_composite/std": 0.348273903131485, "reward": 0.20092102885246277, "reward_std": 0.3482738733291626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030687645077705383, "sampling/sampling_logp_difference/max": 1.2081446647644043, "sampling/importance_sampling_ratio/min": 0.29875102639198303, "sampling/importance_sampling_ratio/mean": 0.9973870515823364, "sampling/importance_sampling_ratio/max": 1.596976637840271, "entropy": 0.14520971663296223, "clip_ratio/low_mean": 0.01713701686821878, "clip_ratio/low_min": 0.01713701686821878, "clip_ratio/high_mean": 0.007148692850023508, "clip_ratio/high_max": 0.007148692850023508, "clip_ratio/region_mean": 0.024285709718242288, "reward_total_mean": 0.20092102885246277, "reward_meter_mean": 0.20092102885246277, "reward_meter_std": 0.348273903131485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.20092102885246277, "reward_total_composite_std": 0.348273903131485, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1236.0} {"timestamp_utc": "2026-04-11T22:01:36Z", "mode": "train", "global_step": 1237, "epoch": 0.04776799505715169, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.254545454545455e-06, "num_tokens": 2664151.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0007226307061500847, "sampling/sampling_logp_difference/max": 0.11649372428655624, "sampling/importance_sampling_ratio/min": 0.8900356888771057, "sampling/importance_sampling_ratio/mean": 0.9997232556343079, "sampling/importance_sampling_ratio/max": 1.020443320274353, "entropy": 0.00494377754512243, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1237.0} {"timestamp_utc": "2026-04-11T22:01:41Z", "mode": "train", "global_step": 1238, "epoch": 0.04780661105962311, "loss": 0.0009, "grad_norm": 0.4934018850326538, "learning_rate": 6.251515151515152e-06, "num_tokens": 2665942.0, "completions/mean_length": 71.875, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.875, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9897432327270508, "rewards/meter/std": 2.0272691472200677e-05, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4948716163635254, "rewards/total_composite/std": 1.0136345736100338e-05, "reward": 0.4948716163635254, "reward_std": 1.0139330697711557e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00839180313050747, "sampling/sampling_logp_difference/max": 1.5300078392028809, "sampling/importance_sampling_ratio/min": 0.21653397381305695, "sampling/importance_sampling_ratio/mean": 0.9967517852783203, "sampling/importance_sampling_ratio/max": 1.4927079677581787, "entropy": 0.00825893649016507, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0017605633474886417, "clip_ratio/high_max": 0.0017605633474886417, "clip_ratio/region_mean": 0.0017605633474886417, "reward_total_mean": 0.4948716163635254, "reward_meter_mean": 0.9897432327270508, "reward_meter_std": 2.0272691472200677e-05, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4948716163635254, "reward_total_composite_std": 1.0136345736100338e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1238.0} {"timestamp_utc": "2026-04-11T22:01:45Z", "mode": "train", "global_step": 1239, "epoch": 0.047845227062094535, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.248484848484849e-06, "num_tokens": 2667430.0, "completions/mean_length": 37.0, "completions/min_length": 37.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.998460054397583, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.998460054397583, "rewards/total_composite/std": 0.0, "reward": 0.998460054397583, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0004461147473193705, "sampling/sampling_logp_difference/max": 0.017527664080262184, "sampling/importance_sampling_ratio/min": 0.9913299083709717, "sampling/importance_sampling_ratio/mean": 1.0003437995910645, "sampling/importance_sampling_ratio/max": 1.0176821947097778, "entropy": 0.0046099630708340555, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.998460054397583, "reward_meter_mean": 0.998460054397583, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.998460054397583, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1239.0} {"timestamp_utc": "2026-04-11T22:01:50Z", "mode": "train", "global_step": 1240, "epoch": 0.04788384306456596, "loss": 0.0001, "grad_norm": 0.12589673697948456, "learning_rate": 6.245454545454545e-06, "num_tokens": 2669253.0, "completions/mean_length": 72.875, "completions/min_length": 72.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9984616041183472, "rewards/meter/std": 4.404353148856899e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984616041183472, "rewards/total_composite/std": 4.404353148856899e-06, "reward": 0.9984616041183472, "reward_std": 4.3953264139418025e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011335760354995728, "sampling/sampling_logp_difference/max": 5.212934970855713, "sampling/importance_sampling_ratio/min": 0.005445667542517185, "sampling/importance_sampling_ratio/mean": 0.9977725148200989, "sampling/importance_sampling_ratio/max": 1.07175874710083, "entropy": 0.006535896929563023, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9984616041183472, "reward_meter_mean": 0.9984616041183472, "reward_meter_std": 4.404353148856899e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984616041183472, "reward_total_composite_std": 4.404353148856899e-06, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1240.0} {"timestamp_utc": "2026-04-11T22:01:57Z", "mode": "train", "global_step": 1241, "epoch": 0.04792245906703738, "loss": -0.0005, "grad_norm": 0.009827454574406147, "learning_rate": 6.2424242424242434e-06, "num_tokens": 2672876.0, "completions/mean_length": 234.875, "completions/min_length": 234.0, "completions/max_length": 235.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 234.875, "completions/min_terminated_length": 234.0, "completions/max_terminated_length": 235.0, "rewards/meter/mean": 0.998440682888031, "rewards/meter/std": 5.4790903959656134e-05, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7987525463104248, "rewards/total_composite/std": 4.3832722440129146e-05, "reward": 0.7987525463104248, "reward_std": 4.3832722440129146e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00023325848451349884, "sampling/sampling_logp_difference/max": 0.11108684539794922, "sampling/importance_sampling_ratio/min": 0.8948610424995422, "sampling/importance_sampling_ratio/mean": 1.0000439882278442, "sampling/importance_sampling_ratio/max": 1.0212265253067017, "entropy": 0.0013448380414047278, "clip_ratio/low_mean": 0.0005341880605556071, "clip_ratio/low_min": 0.0005341880605556071, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0005341880605556071, "reward_total_mean": 0.7987525463104248, "reward_meter_mean": 0.998440682888031, "reward_meter_std": 5.4790903959656134e-05, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7987525463104248, "reward_total_composite_std": 4.3832722440129146e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1241.0} {"timestamp_utc": "2026-04-11T22:02:02Z", "mode": "train", "global_step": 1242, "epoch": 0.04796107506950881, "loss": 0.046, "grad_norm": 5.925769329071045, "learning_rate": 6.23939393939394e-06, "num_tokens": 2674374.0, "completions/mean_length": 39.25, "completions/min_length": 36.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.2430826723575592, "rewards/meter/std": 0.34168893098831177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.2430826723575592, "rewards/total_composite/std": 0.34168893098831177, "reward": 0.2430826723575592, "reward_std": 0.34168893098831177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.059307683259248734, "sampling/sampling_logp_difference/max": 1.4960908889770508, "sampling/importance_sampling_ratio/min": 0.2240041196346283, "sampling/importance_sampling_ratio/mean": 0.9967235326766968, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17740085069090128, "clip_ratio/low_mean": 0.049161585280671716, "clip_ratio/low_min": 0.049161585280671716, "clip_ratio/high_mean": 0.020032051717862487, "clip_ratio/high_max": 0.020032051717862487, "clip_ratio/region_mean": 0.0691936369985342, "reward_total_mean": 0.2430826723575592, "reward_meter_mean": 0.2430826723575592, "reward_meter_std": 0.34168893098831177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.2430826723575592, "reward_total_composite_std": 0.34168893098831177, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1242.0} {"timestamp_utc": "2026-04-11T22:02:06Z", "mode": "train", "global_step": 1243, "epoch": 0.04799969107198023, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.236363636363637e-06, "num_tokens": 2675870.0, "completions/mean_length": 24.0, "completions/min_length": 24.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.0, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.990179181098938, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.990179181098938, "rewards/total_composite/std": 0.0, "reward": 0.990179181098938, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.001943886512890458, "sampling/sampling_logp_difference/max": 0.1477702409029007, "sampling/importance_sampling_ratio/min": 0.8626292943954468, "sampling/importance_sampling_ratio/mean": 1.0003488063812256, "sampling/importance_sampling_ratio/max": 1.0428940057754517, "entropy": 0.01078690349822864, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.990179181098938, "reward_meter_mean": 0.990179181098938, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.990179181098938, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1243.0} {"timestamp_utc": "2026-04-11T22:02:12Z", "mode": "train", "global_step": 1244, "epoch": 0.048038307074451655, "loss": -0.0196, "grad_norm": 1.7217738628387451, "learning_rate": 6.2333333333333335e-06, "num_tokens": 2678620.0, "completions/mean_length": 143.75, "completions/min_length": 124.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.75, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.014318069443106651, "rewards/meter/std": 0.002862258581444621, "rewards/count_adherence/mean": 0.6000000238418579, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.00859084166586399, "rewards/total_composite/std": 0.001717355102300644, "reward": 0.00859084166586399, "reward_std": 0.0017173553351312876, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011793490499258041, "sampling/sampling_logp_difference/max": 0.7456154823303223, "sampling/importance_sampling_ratio/min": 0.4744422137737274, "sampling/importance_sampling_ratio/mean": 1.0002766847610474, "sampling/importance_sampling_ratio/max": 1.551698088645935, "entropy": 0.06293774722144008, "clip_ratio/low_mean": 0.0010080644860863686, "clip_ratio/low_min": 0.0010080644860863686, "clip_ratio/high_mean": 0.005098244757391512, "clip_ratio/high_max": 0.005098244757391512, "clip_ratio/region_mean": 0.006106309243477881, "reward_total_mean": 0.00859084166586399, "reward_meter_mean": 0.014318069443106651, "reward_meter_std": 0.002862258581444621, "reward_count_adherence_mean": 0.6000000238418579, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.00859084166586399, "reward_total_composite_std": 0.001717355102300644, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1244.0} {"timestamp_utc": "2026-04-11T22:02:17Z", "mode": "train", "global_step": 1245, "epoch": 0.04807692307692308, "loss": 0.0694, "grad_norm": 8.343432426452637, "learning_rate": 6.230303030303031e-06, "num_tokens": 2680323.0, "completions/mean_length": 57.875, "completions/min_length": 53.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.06858856976032257, "rewards/meter/std": 0.03098926693201065, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.26726123690605164, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.05854678526520729, "rewards/total_composite/std": 0.041416432708501816, "reward": 0.05854678526520729, "reward_std": 0.041416432708501816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03031887486577034, "sampling/sampling_logp_difference/max": 1.6382564306259155, "sampling/importance_sampling_ratio/min": 0.19431854784488678, "sampling/importance_sampling_ratio/mean": 0.9979752898216248, "sampling/importance_sampling_ratio/max": 1.822403907775879, "entropy": 0.09279528632760048, "clip_ratio/low_mean": 0.018145160749554634, "clip_ratio/low_min": 0.018145160749554634, "clip_ratio/high_mean": 0.0116177499294281, "clip_ratio/high_max": 0.0116177499294281, "clip_ratio/region_mean": 0.029762910678982735, "reward_total_mean": 0.05854678526520729, "reward_meter_mean": 0.06858856976032257, "reward_meter_std": 0.03098926693201065, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.26726123690605164, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.05854678526520729, "reward_total_composite_std": 0.041416432708501816, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1245.0} {"timestamp_utc": "2026-04-11T22:02:21Z", "mode": "train", "global_step": 1246, "epoch": 0.048115539079394504, "loss": -0.0084, "grad_norm": 2.511941432952881, "learning_rate": 6.227272727272727e-06, "num_tokens": 2681764.0, "completions/mean_length": 24.125, "completions/min_length": 24.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.125, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.9906346797943115, "rewards/meter/std": 0.0012883449671790004, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9906346797943115, "rewards/total_composite/std": 0.0012883449671790004, "reward": 0.9906346797943115, "reward_std": 0.0012883448507636786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004134053830057383, "sampling/sampling_logp_difference/max": 0.4334697723388672, "sampling/importance_sampling_ratio/min": 0.6482558846473694, "sampling/importance_sampling_ratio/mean": 0.999140202999115, "sampling/importance_sampling_ratio/max": 1.1138160228729248, "entropy": 0.018792236573062837, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004999999888241291, "clip_ratio/high_max": 0.004999999888241291, "clip_ratio/region_mean": 0.004999999888241291, "reward_total_mean": 0.9906346797943115, "reward_meter_mean": 0.9906346797943115, "reward_meter_std": 0.0012883449671790004, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9906346797943115, "reward_total_composite_std": 0.0012883449671790004, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1246.0} {"timestamp_utc": "2026-04-11T22:02:28Z", "mode": "train", "global_step": 1247, "epoch": 0.04815415508186593, "loss": -0.015, "grad_norm": 12.830361366271973, "learning_rate": 6.224242424242425e-06, "num_tokens": 2684654.0, "completions/mean_length": 196.25, "completions/min_length": 188.0, "completions/max_length": 199.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 196.25, "completions/min_terminated_length": 188.0, "completions/max_terminated_length": 199.0, "rewards/meter/mean": 0.7086209058761597, "rewards/meter/std": 0.0474582239985466, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6691378355026245, "rewards/total_composite/std": 0.12056674063205719, "reward": 0.6691378355026245, "reward_std": 0.12056674808263779, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0016634892672300339, "sampling/sampling_logp_difference/max": 0.8092460632324219, "sampling/importance_sampling_ratio/min": 0.4451936185359955, "sampling/importance_sampling_ratio/mean": 0.999248206615448, "sampling/importance_sampling_ratio/max": 1.1148637533187866, "entropy": 0.005303148180246353, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0012562813935801387, "clip_ratio/high_max": 0.0012562813935801387, "clip_ratio/region_mean": 0.0012562813935801387, "reward_total_mean": 0.6691378355026245, "reward_meter_mean": 0.7086209058761597, "reward_meter_std": 0.0474582239985466, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6691378355026245, "reward_total_composite_std": 0.12056674063205719, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1247.0} {"timestamp_utc": "2026-04-11T22:02:33Z", "mode": "train", "global_step": 1248, "epoch": 0.04819277108433735, "loss": 0.0632, "grad_norm": 4.916164398193359, "learning_rate": 6.221212121212121e-06, "num_tokens": 2686486.0, "completions/mean_length": 59.0, "completions/min_length": 55.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.08030078560113907, "rewards/meter/std": 0.02289426326751709, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.07489196956157684, "rewards/total_composite/std": 0.03289766609668732, "reward": 0.07489196956157684, "reward_std": 0.03289766609668732, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021717576310038567, "sampling/sampling_logp_difference/max": 1.3160091638565063, "sampling/importance_sampling_ratio/min": 0.2682035267353058, "sampling/importance_sampling_ratio/mean": 0.9987189173698425, "sampling/importance_sampling_ratio/max": 1.4320318698883057, "entropy": 0.08454679185524583, "clip_ratio/low_mean": 0.005741003900766373, "clip_ratio/low_min": 0.005741003900766373, "clip_ratio/high_mean": 0.011243307264521718, "clip_ratio/high_max": 0.011243307264521718, "clip_ratio/region_mean": 0.01698431116528809, "reward_total_mean": 0.07489196956157684, "reward_meter_mean": 0.08030078560113907, "reward_meter_std": 0.02289426326751709, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.07489196956157684, "reward_total_composite_std": 0.03289766609668732, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1248.0} {"timestamp_utc": "2026-04-11T22:02:38Z", "mode": "train", "global_step": 1249, "epoch": 0.048231387086808776, "loss": -0.0003, "grad_norm": 22.853866577148438, "learning_rate": 6.218181818181819e-06, "num_tokens": 2688005.0, "completions/mean_length": 28.875, "completions/min_length": 27.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.5867348909378052, "rewards/meter/std": 0.34122809767723083, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5867348909378052, "rewards/total_composite/std": 0.34122809767723083, "reward": 0.5867348909378052, "reward_std": 0.34122809767723083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07755798101425171, "sampling/sampling_logp_difference/max": 7.22206449508667, "sampling/importance_sampling_ratio/min": 0.0007302931626327336, "sampling/importance_sampling_ratio/mean": 0.9829212427139282, "sampling/importance_sampling_ratio/max": 1.9198836088180542, "entropy": 0.1876389365643263, "clip_ratio/low_mean": 0.013723545242100954, "clip_ratio/low_min": 0.013723545242100954, "clip_ratio/high_mean": 0.03331704158335924, "clip_ratio/high_max": 0.03331704158335924, "clip_ratio/region_mean": 0.047040586825460196, "reward_total_mean": 0.5867348909378052, "reward_meter_mean": 0.5867348909378052, "reward_meter_std": 0.34122809767723083, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5867348909378052, "reward_total_composite_std": 0.34122809767723083, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1249.0} {"timestamp_utc": "2026-04-11T22:02:43Z", "mode": "train", "global_step": 1250, "epoch": 0.0482700030892802, "loss": 0.0111, "grad_norm": 6.257474899291992, "learning_rate": 6.215151515151515e-06, "num_tokens": 2689743.0, "completions/mean_length": 57.25, "completions/min_length": 49.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.25, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.28919148445129395, "rewards/meter/std": 0.3240744471549988, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.28919148445129395, "rewards/total_composite/std": 0.3240744471549988, "reward": 0.28919148445129395, "reward_std": 0.3240744173526764, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0918583795428276, "sampling/sampling_logp_difference/max": 7.61323356628418, "sampling/importance_sampling_ratio/min": 0.0004938722704537213, "sampling/importance_sampling_ratio/mean": 0.9910808801651001, "sampling/importance_sampling_ratio/max": 1.7948583364486694, "entropy": 0.24437552690505981, "clip_ratio/low_mean": 0.03145771333947778, "clip_ratio/low_min": 0.03145771333947778, "clip_ratio/high_mean": 0.02657365659251809, "clip_ratio/high_max": 0.02657365659251809, "clip_ratio/region_mean": 0.05803136993199587, "reward_total_mean": 0.28919148445129395, "reward_meter_mean": 0.28919148445129395, "reward_meter_std": 0.3240744471549988, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.28919148445129395, "reward_total_composite_std": 0.3240744471549988, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1250.0} {"timestamp_utc": "2026-04-11T22:04:14Z", "mode": "eval", "global_step": 1250, "epoch": 0.0482700030892802, "eval_loss": NaN, "eval_runtime": 90.9365, "eval_samples_per_second": 1.144, "eval_steps_per_second": 0.143, "eval_num_tokens": 2689743.0, "eval_completions/mean_length": 272.21153846153845, "eval_completions/min_length": 67.46153846153847, "eval_completions/max_length": 490.84615384615387, "eval_completions/clipped_ratio": 0.17307692307692307, "eval_completions/mean_terminated_length": 223.49405494103064, "eval_completions/min_terminated_length": 67.46153846153847, "eval_completions/max_terminated_length": 415.61538461538464, "eval_rewards/meter/mean": 0.48019003638854396, "eval_rewards/meter/std": 0.4327165255179772, "eval_rewards/count_adherence/mean": 0.6980713330782377, "eval_rewards/count_adherence/std": 0.22586544316548568, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/total_composite/mean": 0.37171143981126636, "eval_rewards/total_composite/std": 0.346037195279048, "eval_reward": 0.37171143981126636, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0028765430459036278, "eval_sampling/sampling_logp_difference/max": 0.7452325591674218, "eval_sampling/importance_sampling_ratio/min": 0.4942314945734464, "eval_sampling/importance_sampling_ratio/mean": 1.0001652974348803, "eval_sampling/importance_sampling_ratio/max": 1.224672033236577, "eval_entropy": 0.020595659000369217, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.37171143981126636, "eval_reward_meter_mean": 0.48019003638854396, "eval_reward_meter_std": 0.4327165255179772, "eval_reward_count_adherence_mean": 0.6980713330782377, "eval_reward_count_adherence_std": 0.22586544316548568, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_total_composite_mean": 0.37171143981126636, "eval_reward_total_composite_std": 0.346037195279048, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1250.0} {"timestamp_utc": "2026-04-11T22:04:26Z", "mode": "train", "global_step": 1251, "epoch": 0.048308619091751624, "loss": 0.0046, "grad_norm": 3.153973340988159, "learning_rate": 6.212121212121213e-06, "num_tokens": 2694658.0, "completions/mean_length": 410.375, "completions/min_length": 394.0, "completions/max_length": 422.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 410.375, "completions/min_terminated_length": 394.0, "completions/max_terminated_length": 422.0, "rewards/meter/mean": 0.9545594453811646, "rewards/meter/std": 0.06442203372716904, "rewards/count_adherence/mean": 0.5875000357627869, "rewards/count_adherence/std": 0.0353553481400013, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5627256631851196, "rewards/total_composite/std": 0.06638102978467941, "reward": 0.5627256631851196, "reward_std": 0.06638102233409882, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009992917068302631, "sampling/sampling_logp_difference/max": 2.9213356971740723, "sampling/importance_sampling_ratio/min": 0.053861699998378754, "sampling/importance_sampling_ratio/mean": 0.9976819157600403, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.014585455239284784, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0052173354488331825, "clip_ratio/high_max": 0.0052173354488331825, "clip_ratio/region_mean": 0.0052173354488331825, "reward_total_mean": 0.5627256631851196, "reward_meter_mean": 0.9545594453811646, "reward_meter_std": 0.06442203372716904, "reward_count_adherence_mean": 0.5875000357627869, "reward_count_adherence_std": 0.0353553481400013, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5627256631851196, "reward_total_composite_std": 0.06638102978467941, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1251.0} {"timestamp_utc": "2026-04-11T22:04:31Z", "mode": "train", "global_step": 1252, "epoch": 0.04834723509422305, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.209090909090909e-06, "num_tokens": 2696778.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.998460054397583, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.998460054397583, "rewards/total_composite/std": 0.0, "reward": 0.998460054397583, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005472920020110905, "sampling/sampling_logp_difference/max": 0.022206589579582214, "sampling/importance_sampling_ratio/min": 0.9821140170097351, "sampling/importance_sampling_ratio/mean": 1.000306487083435, "sampling/importance_sampling_ratio/max": 1.0224549770355225, "entropy": 0.004864538263063878, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.998460054397583, "reward_meter_mean": 0.998460054397583, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.998460054397583, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1252.0} {"timestamp_utc": "2026-04-11T22:04:36Z", "mode": "train", "global_step": 1253, "epoch": 0.04838585109669447, "loss": 0.0534, "grad_norm": 3.5526702404022217, "learning_rate": 6.206060606060606e-06, "num_tokens": 2699100.0, "completions/mean_length": 101.25, "completions/min_length": 96.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.25, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.9928406476974487, "rewards/meter/std": 0.0015889843925833702, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.1178511381149292, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6205252408981323, "rewards/total_composite/std": 0.11701169610023499, "reward": 0.6205252408981323, "reward_std": 0.11701168864965439, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019264565780758858, "sampling/sampling_logp_difference/max": 2.357682466506958, "sampling/importance_sampling_ratio/min": 0.09463929384946823, "sampling/importance_sampling_ratio/mean": 0.9954177141189575, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.025713024253491312, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.013777330634184182, "clip_ratio/high_max": 0.013777330634184182, "clip_ratio/region_mean": 0.013777330634184182, "reward_total_mean": 0.6205252408981323, "reward_meter_mean": 0.9928406476974487, "reward_meter_std": 0.0015889843925833702, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.1178511381149292, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6205252408981323, "reward_total_composite_std": 0.11701169610023499, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1253.0} {"timestamp_utc": "2026-04-11T22:04:42Z", "mode": "train", "global_step": 1254, "epoch": 0.0484244670991659, "loss": -0.059, "grad_norm": 4.601353645324707, "learning_rate": 6.203030303030304e-06, "num_tokens": 2700990.0, "completions/mean_length": 79.25, "completions/min_length": 68.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.871437668800354, "rewards/meter/std": 0.17707456648349762, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.871437668800354, "rewards/total_composite/std": 0.17707456648349762, "reward": 0.871437668800354, "reward_std": 0.17707456648349762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033248189836740494, "sampling/sampling_logp_difference/max": 2.104376792907715, "sampling/importance_sampling_ratio/min": 0.12192162871360779, "sampling/importance_sampling_ratio/mean": 0.991167426109314, "sampling/importance_sampling_ratio/max": 1.5579140186309814, "entropy": 0.05939881457015872, "clip_ratio/low_mean": 0.007246376946568489, "clip_ratio/low_min": 0.007246376946568489, "clip_ratio/high_mean": 0.021220930386334658, "clip_ratio/high_max": 0.021220930386334658, "clip_ratio/region_mean": 0.028467307332903147, "reward_total_mean": 0.871437668800354, "reward_meter_mean": 0.871437668800354, "reward_meter_std": 0.17707456648349762, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.871437668800354, "reward_total_composite_std": 0.17707456648349762, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1254.0} {"timestamp_utc": "2026-04-11T22:04:46Z", "mode": "train", "global_step": 1255, "epoch": 0.04846308310163732, "loss": -0.0018, "grad_norm": 1.5308072566986084, "learning_rate": 6.200000000000001e-06, "num_tokens": 2702725.0, "completions/mean_length": 57.875, "completions/min_length": 57.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.09173879027366638, "rewards/meter/std": 4.541849921224639e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.09173879027366638, "rewards/total_composite/std": 4.541849921224639e-05, "reward": 0.09173879027366638, "reward_std": 4.541924863588065e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004303749650716782, "sampling/sampling_logp_difference/max": 0.3453063368797302, "sampling/importance_sampling_ratio/min": 0.7080034017562866, "sampling/importance_sampling_ratio/mean": 1.0008472204208374, "sampling/importance_sampling_ratio/max": 1.1698634624481201, "entropy": 0.026861701859161258, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006465517217293382, "clip_ratio/high_max": 0.006465517217293382, "clip_ratio/region_mean": 0.006465517217293382, "reward_total_mean": 0.09173879027366638, "reward_meter_mean": 0.09173879027366638, "reward_meter_std": 4.541849921224639e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.09173879027366638, "reward_total_composite_std": 4.541849921224639e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1255.0} {"timestamp_utc": "2026-04-11T22:04:51Z", "mode": "train", "global_step": 1256, "epoch": 0.048501699104108745, "loss": -0.0264, "grad_norm": 13.0385103225708, "learning_rate": 6.196969696969698e-06, "num_tokens": 2704161.0, "completions/mean_length": 29.5, "completions/min_length": 26.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.8768116235733032, "rewards/meter/std": 0.055638283491134644, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8768116235733032, "rewards/total_composite/std": 0.055638283491134644, "reward": 0.8768116235733032, "reward_std": 0.05563828721642494, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05982372909784317, "sampling/sampling_logp_difference/max": 3.655561923980713, "sampling/importance_sampling_ratio/min": 0.025846971198916435, "sampling/importance_sampling_ratio/mean": 1.0029194355010986, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13367785420268774, "clip_ratio/low_mean": 0.018733421806246042, "clip_ratio/low_min": 0.018733421806246042, "clip_ratio/high_mean": 0.012500000651925802, "clip_ratio/high_max": 0.012500000651925802, "clip_ratio/region_mean": 0.031233422458171844, "reward_total_mean": 0.8768116235733032, "reward_meter_mean": 0.8768116235733032, "reward_meter_std": 0.055638283491134644, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8768116235733032, "reward_total_composite_std": 0.055638283491134644, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1256.0} {"timestamp_utc": "2026-04-11T22:04:55Z", "mode": "train", "global_step": 1257, "epoch": 0.04854031510658017, "loss": 0.0394, "grad_norm": 21.942916870117188, "learning_rate": 6.1939393939393944e-06, "num_tokens": 2705754.0, "completions/mean_length": 29.125, "completions/min_length": 28.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9382262825965881, "rewards/meter/std": 0.05119846388697624, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9382262825965881, "rewards/total_composite/std": 0.05119846388697624, "reward": 0.9382262825965881, "reward_std": 0.051198456436395645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0378846749663353, "sampling/sampling_logp_difference/max": 2.9224627017974854, "sampling/importance_sampling_ratio/min": 0.053801026195287704, "sampling/importance_sampling_ratio/mean": 0.9964342713356018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05812297575175762, "clip_ratio/low_mean": 0.012365591712296009, "clip_ratio/low_min": 0.012365591712296009, "clip_ratio/high_mean": 0.017703202553093433, "clip_ratio/high_max": 0.017703202553093433, "clip_ratio/region_mean": 0.030068794265389442, "reward_total_mean": 0.9382262825965881, "reward_meter_mean": 0.9382262825965881, "reward_meter_std": 0.05119846388697624, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9382262825965881, "reward_total_composite_std": 0.05119846388697624, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1257.0} {"timestamp_utc": "2026-04-11T22:05:01Z", "mode": "train", "global_step": 1258, "epoch": 0.04857893110905159, "loss": -0.0155, "grad_norm": 2.3975887298583984, "learning_rate": 6.190909090909092e-06, "num_tokens": 2708544.0, "completions/mean_length": 175.75, "completions/min_length": 169.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 175.75, "completions/min_terminated_length": 169.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.6326770782470703, "rewards/meter/std": 0.3400861918926239, "rewards/count_adherence/mean": 0.6000000238418579, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3796062469482422, "rewards/total_composite/std": 0.20405170321464539, "reward": 0.3796062469482422, "reward_std": 0.2040516883134842, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014859342947602272, "sampling/sampling_logp_difference/max": 2.5336427688598633, "sampling/importance_sampling_ratio/min": 0.07936936616897583, "sampling/importance_sampling_ratio/mean": 0.9965988397598267, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.015550831914879382, "clip_ratio/low_mean": 0.001479289960116148, "clip_ratio/low_min": 0.001479289960116148, "clip_ratio/high_mean": 0.007119682908523828, "clip_ratio/high_max": 0.007119682908523828, "clip_ratio/region_mean": 0.008598972868639976, "reward_total_mean": 0.3796062469482422, "reward_meter_mean": 0.6326770782470703, "reward_meter_std": 0.3400861918926239, "reward_count_adherence_mean": 0.6000000238418579, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3796062469482422, "reward_total_composite_std": 0.20405170321464539, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1258.0} {"timestamp_utc": "2026-04-11T22:05:12Z", "mode": "train", "global_step": 1259, "epoch": 0.04861754711152302, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.187878787878788e-06, "num_tokens": 2710240.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9764326810836792, "rewards/meter/std": 0.021364793181419373, "rewards/count_adherence/mean": 0.8552631735801697, "rewards/count_adherence/std": 0.08784452825784683, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.83400559425354, "rewards/total_composite/std": 0.07454215735197067, "reward": 0.83400559425354, "reward_std": 0.07454215735197067, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.83400559425354, "reward_meter_mean": 0.9764326810836792, "reward_meter_std": 0.021364793181419373, "reward_count_adherence_mean": 0.8552631735801697, "reward_count_adherence_std": 0.08784452825784683, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.83400559425354, "reward_total_composite_std": 0.07454215735197067, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1259.0} {"timestamp_utc": "2026-04-11T22:05:22Z", "mode": "train", "global_step": 1260, "epoch": 0.04865616311399444, "loss": 0.0083, "grad_norm": 0.760658323764801, "learning_rate": 6.184848484848485e-06, "num_tokens": 2715532.0, "completions/mean_length": 482.5, "completions/min_length": 472.0, "completions/max_length": 485.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 482.5, "completions/min_terminated_length": 472.0, "completions/max_terminated_length": 485.0, "rewards/meter/mean": 0.7931843996047974, "rewards/meter/std": 0.00018872729560825974, "rewards/count_adherence/mean": 0.5803571939468384, "rewards/count_adherence/std": 0.02525380253791809, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4603334069252014, "rewards/total_composite/std": 0.020121604204177856, "reward": 0.4603334069252014, "reward_std": 0.020121607929468155, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0013607984874397516, "sampling/sampling_logp_difference/max": 0.7859268188476562, "sampling/importance_sampling_ratio/min": 0.4556971788406372, "sampling/importance_sampling_ratio/mean": 0.9998324513435364, "sampling/importance_sampling_ratio/max": 1.489126205444336, "entropy": 0.0027480133867356926, "clip_ratio/low_mean": 0.001033057807944715, "clip_ratio/low_min": 0.001033057807944715, "clip_ratio/high_mean": 0.00026483050896786153, "clip_ratio/high_max": 0.00026483050896786153, "clip_ratio/region_mean": 0.0012978883169125766, "reward_total_mean": 0.4603334069252014, "reward_meter_mean": 0.7931843996047974, "reward_meter_std": 0.00018872729560825974, "reward_count_adherence_mean": 0.5803571939468384, "reward_count_adherence_std": 0.02525380253791809, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4603334069252014, "reward_total_composite_std": 0.020121604204177856, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1260.0} {"timestamp_utc": "2026-04-11T22:05:28Z", "mode": "train", "global_step": 1261, "epoch": 0.048694779116465865, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.181818181818182e-06, "num_tokens": 2718140.0, "completions/mean_length": 148.0, "completions/min_length": 148.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 148.0, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.8144195675849915, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4072097837924957, "rewards/total_composite/std": 0.0, "reward": 0.4072097837924957, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00014861102681607008, "sampling/sampling_logp_difference/max": 0.021021300926804543, "sampling/importance_sampling_ratio/min": 0.9939432740211487, "sampling/importance_sampling_ratio/mean": 1.0001336336135864, "sampling/importance_sampling_ratio/max": 1.0212438106536865, "entropy": 0.001060088659869507, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.4072097837924957, "reward_meter_mean": 0.8144195675849915, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4072097837924957, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1261.0} {"timestamp_utc": "2026-04-11T22:05:36Z", "mode": "train", "global_step": 1262, "epoch": 0.04873339511893729, "loss": 0.0173, "grad_norm": 3.1900687217712402, "learning_rate": 6.17878787878788e-06, "num_tokens": 2722020.0, "completions/mean_length": 289.0, "completions/min_length": 280.0, "completions/max_length": 292.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 289.0, "completions/min_terminated_length": 280.0, "completions/max_terminated_length": 292.0, "rewards/meter/mean": 0.7994517087936401, "rewards/meter/std": 0.0006388832698576152, "rewards/count_adherence/mean": 0.6944444179534912, "rewards/count_adherence/std": 0.05143444612622261, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5552035570144653, "rewards/total_composite/std": 0.04159851744771004, "reward": 0.5552035570144653, "reward_std": 0.04159851372241974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00387288024649024, "sampling/sampling_logp_difference/max": 5.066432476043701, "sampling/importance_sampling_ratio/min": 0.006304872687906027, "sampling/importance_sampling_ratio/mean": 0.9994969367980957, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0011521001142682508, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.00044642857392318547, "clip_ratio/high_max": 0.00044642857392318547, "clip_ratio/region_mean": 0.00044642857392318547, "reward_total_mean": 0.5552035570144653, "reward_meter_mean": 0.7994517087936401, "reward_meter_std": 0.0006388832698576152, "reward_count_adherence_mean": 0.6944444179534912, "reward_count_adherence_std": 0.05143444612622261, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5552035570144653, "reward_total_composite_std": 0.04159851744771004, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1262.0} {"timestamp_utc": "2026-04-11T22:05:41Z", "mode": "train", "global_step": 1263, "epoch": 0.048772011121408713, "loss": 0.001, "grad_norm": 2.1536705493927, "learning_rate": 6.175757575757576e-06, "num_tokens": 2724545.0, "completions/mean_length": 123.625, "completions/min_length": 123.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.625, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9715100526809692, "rewards/meter/std": 0.014701100066304207, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6476733684539795, "rewards/total_composite/std": 0.009800735861063004, "reward": 0.6476733684539795, "reward_std": 0.009800727479159832, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00393439456820488, "sampling/sampling_logp_difference/max": 1.301499843597412, "sampling/importance_sampling_ratio/min": 0.2721233367919922, "sampling/importance_sampling_ratio/mean": 0.9988226890563965, "sampling/importance_sampling_ratio/max": 1.2851040363311768, "entropy": 0.007236705539980903, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0010162601247429848, "clip_ratio/high_max": 0.0010162601247429848, "clip_ratio/region_mean": 0.0010162601247429848, "reward_total_mean": 0.6476733684539795, "reward_meter_mean": 0.9715100526809692, "reward_meter_std": 0.014701100066304207, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6476733684539795, "reward_total_composite_std": 0.009800735861063004, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1263.0} {"timestamp_utc": "2026-04-11T22:05:47Z", "mode": "train", "global_step": 1264, "epoch": 0.04881062712388014, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.1727272727272735e-06, "num_tokens": 2726681.0, "completions/mean_length": 100.0, "completions/min_length": 100.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.0, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.8300259113311768, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5533506274223328, "rewards/total_composite/std": 0.0, "reward": 0.5533506274223328, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 4.091426308150403e-05, "sampling/sampling_logp_difference/max": 0.0010204353602603078, "sampling/importance_sampling_ratio/min": 0.9989801049232483, "sampling/importance_sampling_ratio/mean": 1.000034213066101, "sampling/importance_sampling_ratio/max": 1.0010144710540771, "entropy": 0.0008837012210278772, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.5533506274223328, "reward_meter_mean": 0.8300259113311768, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5533506274223328, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1264.0} {"timestamp_utc": "2026-04-11T22:05:52Z", "mode": "train", "global_step": 1265, "epoch": 0.04884924312635156, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.16969696969697e-06, "num_tokens": 2728817.0, "completions/mean_length": 92.0, "completions/min_length": 92.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.0, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.46078968048095703, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3071931302547455, "rewards/total_composite/std": 0.0, "reward": 0.3071931302547455, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0016213931376114488, "sampling/sampling_logp_difference/max": 0.40386709570884705, "sampling/importance_sampling_ratio/min": 0.667732834815979, "sampling/importance_sampling_ratio/mean": 0.999958872795105, "sampling/importance_sampling_ratio/max": 1.0460588932037354, "entropy": 0.009082602395210415, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.3071931302547455, "reward_meter_mean": 0.46078968048095703, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3071931302547455, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1265.0} {"timestamp_utc": "2026-04-11T22:05:57Z", "mode": "train", "global_step": 1266, "epoch": 0.048887859128822986, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.166666666666667e-06, "num_tokens": 2731033.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.998460054397583, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.998460054397583, "rewards/total_composite/std": 0.0, "reward": 0.998460054397583, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0014422583626583219, "sampling/sampling_logp_difference/max": 0.27352064847946167, "sampling/importance_sampling_ratio/min": 0.7606965899467468, "sampling/importance_sampling_ratio/mean": 0.9996913075447083, "sampling/importance_sampling_ratio/max": 1.0756375789642334, "entropy": 0.008182921446859837, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.998460054397583, "reward_meter_mean": 0.998460054397583, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.998460054397583, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1266.0} {"timestamp_utc": "2026-04-11T22:06:01Z", "mode": "train", "global_step": 1267, "epoch": 0.04892647513129441, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.163636363636364e-06, "num_tokens": 2732473.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9877970814704895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9877970814704895, "rewards/total_composite/std": 0.0, "reward": 0.9877970814704895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0014713035197928548, "sampling/sampling_logp_difference/max": 0.04749540984630585, "sampling/importance_sampling_ratio/min": 0.9536148905754089, "sampling/importance_sampling_ratio/mean": 1.0006672143936157, "sampling/importance_sampling_ratio/max": 1.0476287603378296, "entropy": 0.011590654379688203, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9877970814704895, "reward_meter_mean": 0.9877970814704895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9877970814704895, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1267.0} {"timestamp_utc": "2026-04-11T22:06:06Z", "mode": "train", "global_step": 1268, "epoch": 0.048965091133765834, "loss": -0.0001, "grad_norm": 2.2302300930023193, "learning_rate": 6.160606060606062e-06, "num_tokens": 2734465.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.8916593790054321, "rewards/meter/std": 0.30102142691612244, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.44582968950271606, "rewards/total_composite/std": 0.15051071345806122, "reward": 0.44582968950271606, "reward_std": 0.15051071345806122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009842847473919392, "sampling/sampling_logp_difference/max": 1.490877628326416, "sampling/importance_sampling_ratio/min": 0.2251749485731125, "sampling/importance_sampling_ratio/mean": 1.0005149841308594, "sampling/importance_sampling_ratio/max": 1.7311148643493652, "entropy": 0.02477679366711527, "clip_ratio/low_mean": 0.004120879340916872, "clip_ratio/low_min": 0.004120879340916872, "clip_ratio/high_mean": 0.00412087922450155, "clip_ratio/high_max": 0.00412087922450155, "clip_ratio/region_mean": 0.008241758565418422, "reward_total_mean": 0.44582968950271606, "reward_meter_mean": 0.8916593790054321, "reward_meter_std": 0.30102142691612244, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.44582968950271606, "reward_total_composite_std": 0.15051071345806122, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1268.0} {"timestamp_utc": "2026-04-11T22:06:16Z", "mode": "train", "global_step": 1269, "epoch": 0.04900370713623726, "loss": -0.0964, "grad_norm": 5.119290351867676, "learning_rate": 6.157575757575758e-06, "num_tokens": 2736626.0, "completions/mean_length": 146.125, "completions/min_length": 85.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 93.85714721679688, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.40928274393081665, "rewards/meter/std": 0.2772928774356842, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.1178511381149292, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.27111589908599854, "rewards/total_composite/std": 0.18769000470638275, "reward": 0.27111589908599854, "reward_std": 0.18768998980522156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04919489845633507, "sampling/sampling_logp_difference/max": 1.7027902603149414, "sampling/importance_sampling_ratio/min": 0.18217450380325317, "sampling/importance_sampling_ratio/mean": 0.9988682270050049, "sampling/importance_sampling_ratio/max": 1.7967910766601562, "entropy": 0.18736353144049644, "clip_ratio/low_mean": 0.024248301284387708, "clip_ratio/low_min": 0.024248301284387708, "clip_ratio/high_mean": 0.020179738756269217, "clip_ratio/high_max": 0.020179738756269217, "clip_ratio/region_mean": 0.044428040040656924, "reward_total_mean": 0.27111589908599854, "reward_meter_mean": 0.40928274393081665, "reward_meter_std": 0.2772928774356842, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.1178511381149292, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.27111589908599854, "reward_total_composite_std": 0.18769000470638275, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1269.0} {"timestamp_utc": "2026-04-11T22:06:22Z", "mode": "train", "global_step": 1270, "epoch": 0.04904232313870868, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.154545454545455e-06, "num_tokens": 2738754.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.49929603934288025, "rewards/total_composite/std": 0.0, "reward": 0.49929603934288025, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 8.755446469876915e-05, "sampling/sampling_logp_difference/max": 0.006327272858470678, "sampling/importance_sampling_ratio/min": 0.9936926960945129, "sampling/importance_sampling_ratio/mean": 1.0000392198562622, "sampling/importance_sampling_ratio/max": 1.0029910802841187, "entropy": 0.0007641121082997415, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.49929603934288025, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.49929603934288025, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1270.0} {"timestamp_utc": "2026-04-11T22:06:27Z", "mode": "train", "global_step": 1271, "epoch": 0.049080939141180106, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.151515151515152e-06, "num_tokens": 2740538.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8786622881889343, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8786622881889343, "rewards/total_composite/std": 0.0, "reward": 0.8786622881889343, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 8.413597970502451e-05, "sampling/sampling_logp_difference/max": 0.0030292102601379156, "sampling/importance_sampling_ratio/min": 0.9969754219055176, "sampling/importance_sampling_ratio/mean": 1.0000576972961426, "sampling/importance_sampling_ratio/max": 1.0026161670684814, "entropy": 0.0007092852247296833, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8786622881889343, "reward_meter_mean": 0.8786622881889343, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8786622881889343, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1271.0} {"timestamp_utc": "2026-04-11T22:06:33Z", "mode": "train", "global_step": 1272, "epoch": 0.04911955514365153, "loss": 0.0574, "grad_norm": 13.880781173706055, "learning_rate": 6.148484848484849e-06, "num_tokens": 2742436.0, "completions/mean_length": 89.25, "completions/min_length": 82.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.25, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9970121383666992, "rewards/meter/std": 0.0014506883453577757, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9970121383666992, "rewards/total_composite/std": 0.0014506883453577757, "reward": 0.9970121383666992, "reward_std": 0.0014506981242448092, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022777754813432693, "sampling/sampling_logp_difference/max": 2.1660735607147217, "sampling/importance_sampling_ratio/min": 0.11462680995464325, "sampling/importance_sampling_ratio/mean": 1.0056332349777222, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07023449242115021, "clip_ratio/low_mean": 0.006739542470313609, "clip_ratio/low_min": 0.006739542470313609, "clip_ratio/high_mean": 0.008869817247614264, "clip_ratio/high_max": 0.008869817247614264, "clip_ratio/region_mean": 0.015609359717927873, "reward_total_mean": 0.9970121383666992, "reward_meter_mean": 0.9970121383666992, "reward_meter_std": 0.0014506883453577757, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9970121383666992, "reward_total_composite_std": 0.0014506883453577757, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1272.0} {"timestamp_utc": "2026-04-11T22:06:39Z", "mode": "train", "global_step": 1273, "epoch": 0.049158171146122955, "loss": -0.0215, "grad_norm": 4.542704105377197, "learning_rate": 6.1454545454545454e-06, "num_tokens": 2744972.0, "completions/mean_length": 130.0, "completions/min_length": 121.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.0, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.29038572311401367, "rewards/meter/std": 0.4377255141735077, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.17996034026145935, "rewards/total_composite/std": 0.28114935755729675, "reward": 0.17996034026145935, "reward_std": 0.28114935755729675, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03621600940823555, "sampling/sampling_logp_difference/max": 1.5051696300506592, "sampling/importance_sampling_ratio/min": 0.22197963297367096, "sampling/importance_sampling_ratio/mean": 0.997532308101654, "sampling/importance_sampling_ratio/max": 1.8310456275939941, "entropy": 0.12695379834622145, "clip_ratio/low_mean": 0.021548386896029115, "clip_ratio/low_min": 0.021548386896029115, "clip_ratio/high_mean": 0.007387349382042885, "clip_ratio/high_max": 0.007387349382042885, "clip_ratio/region_mean": 0.028935736278072, "reward_total_mean": 0.17996034026145935, "reward_meter_mean": 0.29038572311401367, "reward_meter_std": 0.4377255141735077, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.17996034026145935, "reward_total_composite_std": 0.28114935755729675, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1273.0} {"timestamp_utc": "2026-04-11T22:06:45Z", "mode": "train", "global_step": 1274, "epoch": 0.04919678714859438, "loss": 0.0057, "grad_norm": 0.17971758544445038, "learning_rate": 6.142424242424243e-06, "num_tokens": 2748337.0, "completions/mean_length": 196.625, "completions/min_length": 196.0, "completions/max_length": 201.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 196.625, "completions/min_terminated_length": 196.0, "completions/max_terminated_length": 201.0, "rewards/meter/mean": 0.9540700912475586, "rewards/meter/std": 0.12592719495296478, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7632560729980469, "rewards/total_composite/std": 0.10074175894260406, "reward": 0.7632560729980469, "reward_std": 0.10074175894260406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0009647245751693845, "sampling/sampling_logp_difference/max": 0.573521614074707, "sampling/importance_sampling_ratio/min": 0.5635373592376709, "sampling/importance_sampling_ratio/mean": 0.9998683333396912, "sampling/importance_sampling_ratio/max": 1.298904538154602, "entropy": 0.0061359210340015125, "clip_ratio/low_mean": 0.0037313431967049837, "clip_ratio/low_min": 0.0037313431967049837, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0037313431967049837, "reward_total_mean": 0.7632560729980469, "reward_meter_mean": 0.9540700912475586, "reward_meter_std": 0.12592719495296478, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7632560729980469, "reward_total_composite_std": 0.10074175894260406, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1274.0} {"timestamp_utc": "2026-04-11T22:06:51Z", "mode": "train", "global_step": 1275, "epoch": 0.0492354031510658, "loss": 0.0213, "grad_norm": 3.6083242893218994, "learning_rate": 6.139393939393939e-06, "num_tokens": 2750880.0, "completions/mean_length": 157.875, "completions/min_length": 148.0, "completions/max_length": 166.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.875, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.2238028645515442, "rewards/meter/std": 0.16664241254329681, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.149201899766922, "rewards/total_composite/std": 0.11109494417905807, "reward": 0.149201899766922, "reward_std": 0.11109494417905807, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05217120051383972, "sampling/sampling_logp_difference/max": 6.626462936401367, "sampling/importance_sampling_ratio/min": 0.001324840821325779, "sampling/importance_sampling_ratio/mean": 0.9991512894630432, "sampling/importance_sampling_ratio/max": 1.8626716136932373, "entropy": 0.20470203552395105, "clip_ratio/low_mean": 0.01942119817249477, "clip_ratio/low_min": 0.01942119817249477, "clip_ratio/high_mean": 0.016115174745209515, "clip_ratio/high_max": 0.016115174745209515, "clip_ratio/region_mean": 0.035536372917704284, "reward_total_mean": 0.149201899766922, "reward_meter_mean": 0.2238028645515442, "reward_meter_std": 0.16664241254329681, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.149201899766922, "reward_total_composite_std": 0.11109494417905807, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1275.0} {"timestamp_utc": "2026-04-11T22:06:56Z", "mode": "train", "global_step": 1276, "epoch": 0.04927401915353723, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.136363636363637e-06, "num_tokens": 2752816.0, "completions/mean_length": 81.0, "completions/min_length": 81.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.0, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9917338490486145, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.49586692452430725, "rewards/total_composite/std": 0.0, "reward": 0.49586692452430725, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0014557959511876106, "sampling/sampling_logp_difference/max": 0.14818421006202698, "sampling/importance_sampling_ratio/min": 0.9737723469734192, "sampling/importance_sampling_ratio/mean": 1.001383900642395, "sampling/importance_sampling_ratio/max": 1.1597265005111694, "entropy": 0.009700042544864118, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.49586692452430725, "reward_meter_mean": 0.9917338490486145, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.49586692452430725, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1276.0} {"timestamp_utc": "2026-04-11T22:07:06Z", "mode": "train", "global_step": 1277, "epoch": 0.04931263515600865, "loss": -0.1194, "grad_norm": 2.2128188610076904, "learning_rate": 6.133333333333334e-06, "num_tokens": 2755279.0, "completions/mean_length": 201.875, "completions/min_length": 153.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 157.57144165039062, "completions/min_terminated_length": 153.0, "completions/max_terminated_length": 165.0, "rewards/meter/mean": 0.3189830183982849, "rewards/meter/std": 0.2755068242549896, "rewards/count_adherence/mean": 0.5833333730697632, "rewards/count_adherence/std": 0.2357022762298584, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.21265536546707153, "rewards/total_composite/std": 0.183671236038208, "reward": 0.21265536546707153, "reward_std": 0.18367122113704681, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021676570177078247, "sampling/sampling_logp_difference/max": 1.5166254043579102, "sampling/importance_sampling_ratio/min": 0.2194512039422989, "sampling/importance_sampling_ratio/mean": 0.9984681010246277, "sampling/importance_sampling_ratio/max": 1.688173770904541, "entropy": 0.10916282888501883, "clip_ratio/low_mean": 0.014305056189186871, "clip_ratio/low_min": 0.014305056189186871, "clip_ratio/high_mean": 0.0055407010950148106, "clip_ratio/high_max": 0.0055407010950148106, "clip_ratio/region_mean": 0.01984575728420168, "reward_total_mean": 0.21265536546707153, "reward_meter_mean": 0.3189830183982849, "reward_meter_std": 0.2755068242549896, "reward_count_adherence_mean": 0.5833333730697632, "reward_count_adherence_std": 0.2357022762298584, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.21265536546707153, "reward_total_composite_std": 0.183671236038208, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1277.0} {"timestamp_utc": "2026-04-11T22:07:11Z", "mode": "train", "global_step": 1278, "epoch": 0.049351251158480075, "loss": 0.0604, "grad_norm": 10.155717849731445, "learning_rate": 6.130303030303031e-06, "num_tokens": 2757162.0, "completions/mean_length": 70.375, "completions/min_length": 67.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.5206853747367859, "rewards/meter/std": 0.3748941123485565, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.26034268736839294, "rewards/total_composite/std": 0.18744705617427826, "reward": 0.26034268736839294, "reward_std": 0.18744705617427826, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02455786243081093, "sampling/sampling_logp_difference/max": 2.3942952156066895, "sampling/importance_sampling_ratio/min": 0.09123696386814117, "sampling/importance_sampling_ratio/mean": 0.9902991056442261, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03526663058437407, "clip_ratio/low_mean": 0.004991369438357651, "clip_ratio/low_min": 0.004991369438357651, "clip_ratio/high_mean": 0.012977392296306789, "clip_ratio/high_max": 0.012977392296306789, "clip_ratio/region_mean": 0.01796876173466444, "reward_total_mean": 0.26034268736839294, "reward_meter_mean": 0.5206853747367859, "reward_meter_std": 0.3748941123485565, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.26034268736839294, "reward_total_composite_std": 0.18744705617427826, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1278.0} {"timestamp_utc": "2026-04-11T22:07:17Z", "mode": "train", "global_step": 1279, "epoch": 0.0493898671609515, "loss": -0.004, "grad_norm": 0.42939507961273193, "learning_rate": 6.127272727272727e-06, "num_tokens": 2759281.0, "completions/mean_length": 111.875, "completions/min_length": 107.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.875, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.992241621017456, "rewards/meter/std": 4.115639967494644e-05, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6614944338798523, "rewards/total_composite/std": 2.7437603421276435e-05, "reward": 0.6614944338798523, "reward_std": 2.7443624276202172e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018404273316264153, "sampling/sampling_logp_difference/max": 5.891563892364502, "sampling/importance_sampling_ratio/min": 0.0027626529335975647, "sampling/importance_sampling_ratio/mean": 0.9980177879333496, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.006762623612303287, "clip_ratio/low_mean": 0.002209890983067453, "clip_ratio/low_min": 0.002209890983067453, "clip_ratio/high_mean": 0.0020833334419876337, "clip_ratio/high_max": 0.0020833334419876337, "clip_ratio/region_mean": 0.004293224425055087, "reward_total_mean": 0.6614944338798523, "reward_meter_mean": 0.992241621017456, "reward_meter_std": 4.115639967494644e-05, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6614944338798523, "reward_total_composite_std": 2.7437603421276435e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1279.0} {"timestamp_utc": "2026-04-11T22:07:22Z", "mode": "train", "global_step": 1280, "epoch": 0.04942848316342292, "loss": 0.0302, "grad_norm": 62.453826904296875, "learning_rate": 6.1242424242424245e-06, "num_tokens": 2761030.0, "completions/mean_length": 61.625, "completions/min_length": 61.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9895129799842834, "rewards/meter/std": 0.022948207333683968, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9895129799842834, "rewards/total_composite/std": 0.022948207333683968, "reward": 0.9895129799842834, "reward_std": 0.022948196157813072, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011571628972887993, "sampling/sampling_logp_difference/max": 0.791548490524292, "sampling/importance_sampling_ratio/min": 0.45314255356788635, "sampling/importance_sampling_ratio/mean": 0.9998781681060791, "sampling/importance_sampling_ratio/max": 1.8315465450286865, "entropy": 0.04697659378871322, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006114489398896694, "clip_ratio/high_max": 0.006114489398896694, "clip_ratio/region_mean": 0.006114489398896694, "reward_total_mean": 0.9895129799842834, "reward_meter_mean": 0.9895129799842834, "reward_meter_std": 0.022948207333683968, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9895129799842834, "reward_total_composite_std": 0.022948207333683968, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1280.0} {"timestamp_utc": "2026-04-11T22:07:29Z", "mode": "train", "global_step": 1281, "epoch": 0.04946709916589435, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.121212121212121e-06, "num_tokens": 2764502.0, "completions/mean_length": 241.0, "completions/min_length": 241.0, "completions/max_length": 241.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 241.0, "completions/min_terminated_length": 241.0, "completions/max_terminated_length": 241.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 2.3701670215814374e-05, "sampling/sampling_logp_difference/max": 0.002760242437943816, "sampling/importance_sampling_ratio/min": 0.9988345503807068, "sampling/importance_sampling_ratio/mean": 1.000022292137146, "sampling/importance_sampling_ratio/max": 1.002764105796814, "entropy": 0.00018638478468346875, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1281.0} {"timestamp_utc": "2026-04-11T22:07:34Z", "mode": "train", "global_step": 1282, "epoch": 0.04950571516836577, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.118181818181819e-06, "num_tokens": 2766430.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.49929603934288025, "rewards/total_composite/std": 0.0, "reward": 0.49929603934288025, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 4.254432496964e-05, "sampling/sampling_logp_difference/max": 0.0019133060704916716, "sampling/importance_sampling_ratio/min": 0.9993289113044739, "sampling/importance_sampling_ratio/mean": 1.0000381469726562, "sampling/importance_sampling_ratio/max": 1.0019152164459229, "entropy": 0.0003460009320406243, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.49929603934288025, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.49929603934288025, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1282.0} {"timestamp_utc": "2026-04-11T22:07:42Z", "mode": "train", "global_step": 1283, "epoch": 0.049544331170837196, "loss": -0.0062, "grad_norm": 2.424851655960083, "learning_rate": 6.115151515151516e-06, "num_tokens": 2770056.0, "completions/mean_length": 248.25, "completions/min_length": 228.0, "completions/max_length": 258.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 248.25, "completions/min_terminated_length": 228.0, "completions/max_terminated_length": 258.0, "rewards/meter/mean": 0.997593879699707, "rewards/meter/std": 0.0023971654009073973, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.997593879699707, "rewards/total_composite/std": 0.0023971654009073973, "reward": 0.997593879699707, "reward_std": 0.0023971558548510075, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011338972486555576, "sampling/sampling_logp_difference/max": 4.200416088104248, "sampling/importance_sampling_ratio/min": 0.01498933881521225, "sampling/importance_sampling_ratio/mean": 0.9980850219726562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.022822741244453937, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007237150450237095, "clip_ratio/high_max": 0.007237150450237095, "clip_ratio/region_mean": 0.007237150450237095, "reward_total_mean": 0.997593879699707, "reward_meter_mean": 0.997593879699707, "reward_meter_std": 0.0023971654009073973, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.997593879699707, "reward_total_composite_std": 0.0023971654009073973, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1283.0} {"timestamp_utc": "2026-04-11T22:07:47Z", "mode": "train", "global_step": 1284, "epoch": 0.04958294717330862, "loss": 0.0074, "grad_norm": 4.2801079750061035, "learning_rate": 6.112121212121213e-06, "num_tokens": 2772091.0, "completions/mean_length": 76.375, "completions/min_length": 76.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.375, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.01994991861283779, "rewards/meter/std": 0.0197458416223526, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.01994991861283779, "rewards/total_composite/std": 0.0197458416223526, "reward": 0.01994991861283779, "reward_std": 0.01974583975970745, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014959866181015968, "sampling/sampling_logp_difference/max": 1.843103289604187, "sampling/importance_sampling_ratio/min": 0.15832532942295074, "sampling/importance_sampling_ratio/mean": 1.001766324043274, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.028386848978698254, "clip_ratio/low_mean": 0.00978298019617796, "clip_ratio/low_min": 0.00978298019617796, "clip_ratio/high_mean": 0.003289473708719015, "clip_ratio/high_max": 0.003289473708719015, "clip_ratio/region_mean": 0.013072453904896975, "reward_total_mean": 0.01994991861283779, "reward_meter_mean": 0.01994991861283779, "reward_meter_std": 0.0197458416223526, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.01994991861283779, "reward_total_composite_std": 0.0197458416223526, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1284.0} {"timestamp_utc": "2026-04-11T22:07:53Z", "mode": "train", "global_step": 1285, "epoch": 0.049621563175780044, "loss": -0.0594, "grad_norm": 3.203981399536133, "learning_rate": 6.10909090909091e-06, "num_tokens": 2774743.0, "completions/mean_length": 144.5, "completions/min_length": 115.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 144.5, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.01611270383000374, "rewards/meter/std": 0.0017258560983464122, "rewards/count_adherence/mean": 0.6000000238418579, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.009667622856795788, "rewards/total_composite/std": 0.001035513705573976, "reward": 0.009667622856795788, "reward_std": 0.001035513705573976, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02596927061676979, "sampling/sampling_logp_difference/max": 3.5084571838378906, "sampling/importance_sampling_ratio/min": 0.0299430750310421, "sampling/importance_sampling_ratio/mean": 0.9972525238990784, "sampling/importance_sampling_ratio/max": 1.6455178260803223, "entropy": 0.1213388005271554, "clip_ratio/low_mean": 0.0032608695328235626, "clip_ratio/low_min": 0.0032608695328235626, "clip_ratio/high_mean": 0.015982552722562104, "clip_ratio/high_max": 0.015982552722562104, "clip_ratio/region_mean": 0.019243422255385667, "reward_total_mean": 0.009667622856795788, "reward_meter_mean": 0.01611270383000374, "reward_meter_std": 0.0017258560983464122, "reward_count_adherence_mean": 0.6000000238418579, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.009667622856795788, "reward_total_composite_std": 0.001035513705573976, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1285.0} {"timestamp_utc": "2026-04-11T22:07:58Z", "mode": "train", "global_step": 1286, "epoch": 0.04966017917825147, "loss": 0.0049, "grad_norm": 0.8690685033798218, "learning_rate": 6.106060606060606e-06, "num_tokens": 2776611.0, "completions/mean_length": 73.5, "completions/min_length": 61.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9975723624229431, "rewards/meter/std": 4.894016819889657e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9975723624229431, "rewards/total_composite/std": 4.894016819889657e-05, "reward": 0.9975723624229431, "reward_std": 4.893221557722427e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01973308064043522, "sampling/sampling_logp_difference/max": 1.563590168952942, "sampling/importance_sampling_ratio/min": 0.2093830108642578, "sampling/importance_sampling_ratio/mean": 0.9948568344116211, "sampling/importance_sampling_ratio/max": 1.511736512184143, "entropy": 0.046400822000578046, "clip_ratio/low_mean": 0.00805497239343822, "clip_ratio/low_min": 0.00805497239343822, "clip_ratio/high_mean": 0.009328191401436925, "clip_ratio/high_max": 0.009328191401436925, "clip_ratio/region_mean": 0.017383163794875145, "reward_total_mean": 0.9975723624229431, "reward_meter_mean": 0.9975723624229431, "reward_meter_std": 4.894016819889657e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9975723624229431, "reward_total_composite_std": 4.894016819889657e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1286.0} {"timestamp_utc": "2026-04-11T22:08:02Z", "mode": "train", "global_step": 1287, "epoch": 0.04969879518072289, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.103030303030304e-06, "num_tokens": 2778011.0, "completions/mean_length": 28.0, "completions/min_length": 28.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9846518039703369, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9846518039703369, "rewards/total_composite/std": 0.0, "reward": 0.9846518039703369, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0070433709770441055, "sampling/sampling_logp_difference/max": 0.4834262728691101, "sampling/importance_sampling_ratio/min": 0.6166669130325317, "sampling/importance_sampling_ratio/mean": 0.9961096048355103, "sampling/importance_sampling_ratio/max": 1.1136447191238403, "entropy": 0.022762462496757507, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9846518039703369, "reward_meter_mean": 0.9846518039703369, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9846518039703369, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1287.0} {"timestamp_utc": "2026-04-11T22:08:09Z", "mode": "train", "global_step": 1288, "epoch": 0.049737411183194316, "loss": 0.1656, "grad_norm": 1.916330099105835, "learning_rate": 6.1e-06, "num_tokens": 2781143.0, "completions/mean_length": 177.5, "completions/min_length": 124.0, "completions/max_length": 204.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 177.5, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 204.0, "rewards/meter/mean": 0.23511049151420593, "rewards/meter/std": 0.3611241281032562, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.1838536411523819, "rewards/total_composite/std": 0.2662343978881836, "reward": 0.1838536411523819, "reward_std": 0.2662343978881836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009956601075828075, "sampling/sampling_logp_difference/max": 2.1868271827697754, "sampling/importance_sampling_ratio/min": 0.1440984606742859, "sampling/importance_sampling_ratio/mean": 1.0007753372192383, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.009844656538916752, "clip_ratio/low_mean": 0.003846527310088277, "clip_ratio/low_min": 0.003846527310088277, "clip_ratio/high_mean": 0.004000000189989805, "clip_ratio/high_max": 0.004000000189989805, "clip_ratio/region_mean": 0.007846527500078082, "reward_total_mean": 0.1838536411523819, "reward_meter_mean": 0.23511049151420593, "reward_meter_std": 0.3611241281032562, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.1838536411523819, "reward_total_composite_std": 0.2662343978881836, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1288.0} {"timestamp_utc": "2026-04-11T22:08:15Z", "mode": "train", "global_step": 1289, "epoch": 0.04977602718566574, "loss": 0.0258, "grad_norm": 1.954389214515686, "learning_rate": 6.096969696969698e-06, "num_tokens": 2783454.0, "completions/mean_length": 127.875, "completions/min_length": 120.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.875, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9935588240623474, "rewards/meter/std": 0.0004874283040408045, "rewards/count_adherence/mean": 0.375, "rewards/count_adherence/std": 0.1178511381149292, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.37253427505493164, "rewards/total_composite/std": 0.116787388920784, "reward": 0.37253427505493164, "reward_std": 0.1167873814702034, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009776544757187366, "sampling/sampling_logp_difference/max": 3.1669387817382812, "sampling/importance_sampling_ratio/min": 0.04213237762451172, "sampling/importance_sampling_ratio/mean": 0.9972169995307922, "sampling/importance_sampling_ratio/max": 1.2577539682388306, "entropy": 0.008995219715870917, "clip_ratio/low_mean": 0.0009689922444522381, "clip_ratio/low_min": 0.0009689922444522381, "clip_ratio/high_mean": 0.0020833334419876337, "clip_ratio/high_max": 0.0020833334419876337, "clip_ratio/region_mean": 0.003052325686439872, "reward_total_mean": 0.37253427505493164, "reward_meter_mean": 0.9935588240623474, "reward_meter_std": 0.0004874283040408045, "reward_count_adherence_mean": 0.375, "reward_count_adherence_std": 0.1178511381149292, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.37253427505493164, "reward_total_composite_std": 0.116787388920784, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1289.0} {"timestamp_utc": "2026-04-11T22:08:22Z", "mode": "train", "global_step": 1290, "epoch": 0.049814643188137164, "loss": 0.0186, "grad_norm": 1.6063876152038574, "learning_rate": 6.0939393939393946e-06, "num_tokens": 2786265.0, "completions/mean_length": 170.375, "completions/min_length": 165.0, "completions/max_length": 179.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.375, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 179.0, "rewards/meter/mean": 0.012963803485035896, "rewards/meter/std": 0.004894020967185497, "rewards/count_adherence/mean": 0.910714328289032, "rewards/count_adherence/std": 0.07393559068441391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.011587313376367092, "rewards/total_composite/std": 0.004187269136309624, "reward": 0.011587313376367092, "reward_std": 0.004187269136309624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01577018015086651, "sampling/sampling_logp_difference/max": 1.9047927856445312, "sampling/importance_sampling_ratio/min": 0.14885348081588745, "sampling/importance_sampling_ratio/mean": 1.0016361474990845, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06922507751733065, "clip_ratio/low_mean": 0.007179260952398181, "clip_ratio/low_min": 0.007179260952398181, "clip_ratio/high_mean": 0.008912283112294972, "clip_ratio/high_max": 0.008912283112294972, "clip_ratio/region_mean": 0.016091544064693153, "reward_total_mean": 0.011587313376367092, "reward_meter_mean": 0.012963803485035896, "reward_meter_std": 0.004894020967185497, "reward_count_adherence_mean": 0.910714328289032, "reward_count_adherence_std": 0.07393559068441391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.011587313376367092, "reward_total_composite_std": 0.004187269136309624, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1290.0} {"timestamp_utc": "2026-04-11T22:08:28Z", "mode": "train", "global_step": 1291, "epoch": 0.04985325919060859, "loss": 0.0138, "grad_norm": 4.439713954925537, "learning_rate": 6.090909090909092e-06, "num_tokens": 2789481.0, "completions/mean_length": 185.0, "completions/min_length": 184.0, "completions/max_length": 191.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 185.0, "completions/min_terminated_length": 184.0, "completions/max_terminated_length": 191.0, "rewards/meter/mean": 0.012773239985108376, "rewards/meter/std": 0.004585606046020985, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.012773239985108376, "rewards/total_composite/std": 0.004585606046020985, "reward": 0.012773239985108376, "reward_std": 0.004585605580359697, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013530365191400051, "sampling/sampling_logp_difference/max": 4.393998146057129, "sampling/importance_sampling_ratio/min": 0.01235124934464693, "sampling/importance_sampling_ratio/mean": 0.9995017051696777, "sampling/importance_sampling_ratio/max": 1.5968453884124756, "entropy": 0.03622802533209324, "clip_ratio/low_mean": 0.0019633506890386343, "clip_ratio/low_min": 0.0019633506890386343, "clip_ratio/high_mean": 0.004755434871185571, "clip_ratio/high_max": 0.004755434871185571, "clip_ratio/region_mean": 0.006718785560224205, "reward_total_mean": 0.012773239985108376, "reward_meter_mean": 0.012773239985108376, "reward_meter_std": 0.004585606046020985, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.012773239985108376, "reward_total_composite_std": 0.004585606046020985, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1291.0} {"timestamp_utc": "2026-04-11T22:08:36Z", "mode": "train", "global_step": 1292, "epoch": 0.04989187519308001, "loss": -0.004, "grad_norm": 1.4635839462280273, "learning_rate": 6.087878787878788e-06, "num_tokens": 2793140.0, "completions/mean_length": 236.375, "completions/min_length": 230.0, "completions/max_length": 244.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 236.375, "completions/min_terminated_length": 230.0, "completions/max_terminated_length": 244.0, "rewards/meter/mean": 0.01164400763809681, "rewards/meter/std": 0.002892889315262437, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.04629101976752281, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0106651671230793, "rewards/total_composite/std": 0.0023632633965462446, "reward": 0.0106651671230793, "reward_std": 0.002363263163715601, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018551601096987724, "sampling/sampling_logp_difference/max": 5.479066371917725, "sampling/importance_sampling_ratio/min": 0.004173223860561848, "sampling/importance_sampling_ratio/mean": 0.9991459846496582, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04724708991125226, "clip_ratio/low_mean": 0.00374403886962682, "clip_ratio/low_min": 0.00374403886962682, "clip_ratio/high_mean": 0.009520896477624774, "clip_ratio/high_max": 0.009520896477624774, "clip_ratio/region_mean": 0.013264935347251594, "reward_total_mean": 0.0106651671230793, "reward_meter_mean": 0.01164400763809681, "reward_meter_std": 0.002892889315262437, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.04629101976752281, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0106651671230793, "reward_total_composite_std": 0.0023632633965462446, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1292.0} {"timestamp_utc": "2026-04-11T22:08:41Z", "mode": "train", "global_step": 1293, "epoch": 0.04993049119555144, "loss": 0.012, "grad_norm": 2.9739294052124023, "learning_rate": 6.0848484848484855e-06, "num_tokens": 2795202.0, "completions/mean_length": 105.75, "completions/min_length": 103.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.75, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9925870895385742, "rewards/meter/std": 0.000733045453671366, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6617246866226196, "rewards/total_composite/std": 0.0004887010436505079, "reward": 0.6617246866226196, "reward_std": 0.0004886905662715435, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007274853065609932, "sampling/sampling_logp_difference/max": 1.3757386207580566, "sampling/importance_sampling_ratio/min": 0.2526529133319855, "sampling/importance_sampling_ratio/mean": 0.9986278414726257, "sampling/importance_sampling_ratio/max": 1.469567060470581, "entropy": 0.022357701556757092, "clip_ratio/low_mean": 0.004694939125329256, "clip_ratio/low_min": 0.004694939125329256, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004694939125329256, "reward_total_mean": 0.6617246866226196, "reward_meter_mean": 0.9925870895385742, "reward_meter_std": 0.000733045453671366, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6617246866226196, "reward_total_composite_std": 0.0004887010436505079, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1293.0} {"timestamp_utc": "2026-04-11T22:08:46Z", "mode": "train", "global_step": 1294, "epoch": 0.04996910719802286, "loss": 0.0023, "grad_norm": 2.022381544113159, "learning_rate": 6.081818181818182e-06, "num_tokens": 2796986.0, "completions/mean_length": 61.0, "completions/min_length": 60.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9975788593292236, "rewards/meter/std": 0.00013958557974547148, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9975788593292236, "rewards/total_composite/std": 0.00013958557974547148, "reward": 0.9975788593292236, "reward_std": 0.00013957775081507862, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008819255977869034, "sampling/sampling_logp_difference/max": 0.9729304313659668, "sampling/importance_sampling_ratio/min": 0.3779737949371338, "sampling/importance_sampling_ratio/mean": 1.0032329559326172, "sampling/importance_sampling_ratio/max": 1.4240106344223022, "entropy": 0.030725523713044822, "clip_ratio/low_mean": 0.004099462414160371, "clip_ratio/low_min": 0.004099462414160371, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004099462414160371, "reward_total_mean": 0.9975788593292236, "reward_meter_mean": 0.9975788593292236, "reward_meter_std": 0.00013958557974547148, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9975788593292236, "reward_total_composite_std": 0.00013958557974547148, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1294.0} {"timestamp_utc": "2026-04-11T22:08:57Z", "mode": "train", "global_step": 1295, "epoch": 0.050007723200494285, "loss": -0.0433, "grad_norm": 0.4125716984272003, "learning_rate": 6.07878787878788e-06, "num_tokens": 2798075.0, "completions/mean_length": 456.125, "completions/min_length": 65.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.875, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.09849966317415237, "rewards/meter/std": 0.2785991430282593, "rewards/count_adherence/mean": 0.125, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.09849966317415237, "rewards/total_composite/std": 0.2785991430282593, "reward": 0.09849966317415237, "reward_std": 0.2785991132259369, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04152027145028114, "sampling/sampling_logp_difference/max": 1.9488744735717773, "sampling/importance_sampling_ratio/min": 0.14243429899215698, "sampling/importance_sampling_ratio/mean": 0.9782585501670837, "sampling/importance_sampling_ratio/max": 1.0338245630264282, "entropy": 0.005067803431302309, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.003846153849735856, "clip_ratio/high_max": 0.003846153849735856, "clip_ratio/region_mean": 0.003846153849735856, "reward_total_mean": 0.09849966317415237, "reward_meter_mean": 0.09849966317415237, "reward_meter_std": 0.2785991430282593, "reward_count_adherence_mean": 0.125, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.09849966317415237, "reward_total_composite_std": 0.2785991430282593, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1295.0} {"timestamp_utc": "2026-04-11T22:09:01Z", "mode": "train", "global_step": 1296, "epoch": 0.05004633920296571, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.0757575757575755e-06, "num_tokens": 2799483.0, "completions/mean_length": 31.0, "completions/min_length": 31.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00024119533190969378, "sampling/sampling_logp_difference/max": 0.005844452418386936, "sampling/importance_sampling_ratio/min": 0.9998956918716431, "sampling/importance_sampling_ratio/mean": 1.000240445137024, "sampling/importance_sampling_ratio/max": 1.005861520767212, "entropy": 0.0023338989121839404, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1296.0} {"timestamp_utc": "2026-04-11T22:09:07Z", "mode": "train", "global_step": 1297, "epoch": 0.05008495520543713, "loss": -0.0157, "grad_norm": 2.5595953464508057, "learning_rate": 6.072727272727274e-06, "num_tokens": 2801968.0, "completions/mean_length": 114.625, "completions/min_length": 105.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.625, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.017330879345536232, "rewards/meter/std": 0.00611004838719964, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.012998159974813461, "rewards/total_composite/std": 0.004582536872476339, "reward": 0.012998159974813461, "reward_std": 0.004582536406815052, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012006850913167, "sampling/sampling_logp_difference/max": 1.1150527000427246, "sampling/importance_sampling_ratio/min": 0.3278979957103729, "sampling/importance_sampling_ratio/mean": 0.9978765845298767, "sampling/importance_sampling_ratio/max": 1.500685691833496, "entropy": 0.03907254268415272, "clip_ratio/low_mean": 0.005660891416482627, "clip_ratio/low_min": 0.005660891416482627, "clip_ratio/high_mean": 0.003187172464095056, "clip_ratio/high_max": 0.003187172464095056, "clip_ratio/region_mean": 0.008848063880577683, "reward_total_mean": 0.012998159974813461, "reward_meter_mean": 0.017330879345536232, "reward_meter_std": 0.00611004838719964, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.012998159974813461, "reward_total_composite_std": 0.004582536872476339, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1297.0} {"timestamp_utc": "2026-04-11T22:09:13Z", "mode": "train", "global_step": 1298, "epoch": 0.05012357120790856, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.06969696969697e-06, "num_tokens": 2804528.0, "completions/mean_length": 136.0, "completions/min_length": 136.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.0, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 7.549906149506569e-05, "sampling/sampling_logp_difference/max": 0.004365906119346619, "sampling/importance_sampling_ratio/min": 0.9967855215072632, "sampling/importance_sampling_ratio/mean": 1.0000606775283813, "sampling/importance_sampling_ratio/max": 1.0043754577636719, "entropy": 0.0006647409609286115, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1298.0} {"timestamp_utc": "2026-04-11T22:09:19Z", "mode": "train", "global_step": 1299, "epoch": 0.05016218721037998, "loss": -0.0217, "grad_norm": 3.0739307403564453, "learning_rate": 6.066666666666667e-06, "num_tokens": 2807324.0, "completions/mean_length": 162.5, "completions/min_length": 151.0, "completions/max_length": 174.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 162.5, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 174.0, "rewards/meter/mean": 0.14099237322807312, "rewards/meter/std": 0.22282274067401886, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.14099237322807312, "rewards/total_composite/std": 0.22282274067401886, "reward": 0.14099237322807312, "reward_std": 0.22282274067401886, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023463940247893333, "sampling/sampling_logp_difference/max": 2.1874911785125732, "sampling/importance_sampling_ratio/min": 0.1121978834271431, "sampling/importance_sampling_ratio/mean": 0.9983112812042236, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0832954696379602, "clip_ratio/low_mean": 0.006205387238878757, "clip_ratio/low_min": 0.006205387238878757, "clip_ratio/high_mean": 0.002215396845713258, "clip_ratio/high_max": 0.002215396845713258, "clip_ratio/region_mean": 0.008420784084592015, "reward_total_mean": 0.14099237322807312, "reward_meter_mean": 0.14099237322807312, "reward_meter_std": 0.22282274067401886, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.14099237322807312, "reward_total_composite_std": 0.22282274067401886, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1299.0} {"timestamp_utc": "2026-04-11T22:09:26Z", "mode": "train", "global_step": 1300, "epoch": 0.050200803212851405, "loss": 0.0217, "grad_norm": 3.2614729404449463, "learning_rate": 6.063636363636364e-06, "num_tokens": 2810311.0, "completions/mean_length": 182.375, "completions/min_length": 161.0, "completions/max_length": 208.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 182.375, "completions/min_terminated_length": 161.0, "completions/max_terminated_length": 208.0, "rewards/meter/mean": 0.9973541498184204, "rewards/meter/std": 0.0027852789498865604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9973541498184204, "rewards/total_composite/std": 0.0027852789498865604, "reward": 0.9973541498184204, "reward_std": 0.002785272663459182, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018300553783774376, "sampling/sampling_logp_difference/max": 4.229056358337402, "sampling/importance_sampling_ratio/min": 0.014566130004823208, "sampling/importance_sampling_ratio/mean": 0.9987977147102356, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05100402026437223, "clip_ratio/low_mean": 0.007321946904994547, "clip_ratio/low_min": 0.007321946904994547, "clip_ratio/high_mean": 0.010958889964967966, "clip_ratio/high_max": 0.010958889964967966, "clip_ratio/region_mean": 0.018280836869962513, "reward_total_mean": 0.9973541498184204, "reward_meter_mean": 0.9973541498184204, "reward_meter_std": 0.0027852789498865604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9973541498184204, "reward_total_composite_std": 0.0027852789498865604, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1300.0} {"timestamp_utc": "2026-04-11T22:10:54Z", "mode": "eval", "global_step": 1300, "epoch": 0.050200803212851405, "eval_loss": NaN, "eval_runtime": 87.6363, "eval_samples_per_second": 1.187, "eval_steps_per_second": 0.148, "eval_num_tokens": 2810311.0, "eval_completions/mean_length": 238.67307692307693, "eval_completions/min_length": 67.53846153846153, "eval_completions/max_length": 472.7692307692308, "eval_completions/clipped_ratio": 0.14423076923076922, "eval_completions/mean_terminated_length": 185.75549903282752, "eval_completions/min_terminated_length": 67.53846153846153, "eval_completions/max_terminated_length": 351.46153846153845, "eval_rewards/meter/mean": 0.6081146918810331, "eval_rewards/meter/std": 0.3977002650499344, "eval_rewards/count_adherence/mean": 0.8032533159622779, "eval_rewards/count_adherence/std": 0.25854549614282757, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/total_composite/mean": 0.560040302001513, "eval_rewards/total_composite/std": 0.3606582246720791, "eval_reward": 0.560040302001513, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0035750904890636983, "eval_sampling/sampling_logp_difference/max": 0.6103466336543744, "eval_sampling/importance_sampling_ratio/min": 0.5740250899241521, "eval_sampling/importance_sampling_ratio/mean": 1.0007055080853975, "eval_sampling/importance_sampling_ratio/max": 1.2828241403286273, "eval_entropy": 0.02625207080004307, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.560040302001513, "eval_reward_meter_mean": 0.6081146918810331, "eval_reward_meter_std": 0.3977002650499344, "eval_reward_count_adherence_mean": 0.8032533159622779, "eval_reward_count_adherence_std": 0.25854549614282757, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_total_composite_mean": 0.560040302001513, "eval_reward_total_composite_std": 0.3606582246720791, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1300.0} {"timestamp_utc": "2026-04-11T22:11:04Z", "mode": "train", "global_step": 1301, "epoch": 0.05023941921532283, "loss": -0.0144, "grad_norm": 3.6569223403930664, "learning_rate": 6.060606060606061e-06, "num_tokens": 2812396.0, "completions/mean_length": 93.625, "completions/min_length": 87.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.625, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.03461223840713501, "rewards/meter/std": 0.010290334932506084, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.025299495086073875, "rewards/total_composite/std": 0.012260460294783115, "reward": 0.025299495086073875, "reward_std": 0.01226046122610569, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025608433410525322, "sampling/sampling_logp_difference/max": 6.403406620025635, "sampling/importance_sampling_ratio/min": 0.0016559066716581583, "sampling/importance_sampling_ratio/mean": 0.9993347525596619, "sampling/importance_sampling_ratio/max": 1.3644611835479736, "entropy": 0.06671302812173963, "clip_ratio/low_mean": 0.014077028958126903, "clip_ratio/low_min": 0.014077028958126903, "clip_ratio/high_mean": 0.002572287921793759, "clip_ratio/high_max": 0.002572287921793759, "clip_ratio/region_mean": 0.01664931687992066, "reward_total_mean": 0.025299495086073875, "reward_meter_mean": 0.03461223840713501, "reward_meter_std": 0.010290334932506084, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.025299495086073875, "reward_total_composite_std": 0.012260460294783115, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1301.0} {"timestamp_utc": "2026-04-11T22:11:09Z", "mode": "train", "global_step": 1302, "epoch": 0.050278035217794254, "loss": -0.0014, "grad_norm": 0.6236714124679565, "learning_rate": 6.057575757575757e-06, "num_tokens": 2814131.0, "completions/mean_length": 60.875, "completions/min_length": 60.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9976277351379395, "rewards/meter/std": 4.747844286612235e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9976277351379395, "rewards/total_composite/std": 4.747844286612235e-05, "reward": 0.9976277351379395, "reward_std": 4.7481436922680587e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.001671163714490831, "sampling/sampling_logp_difference/max": 0.1869053840637207, "sampling/importance_sampling_ratio/min": 0.9924246072769165, "sampling/importance_sampling_ratio/mean": 1.0016642808914185, "sampling/importance_sampling_ratio/max": 1.2055132389068604, "entropy": 0.009936069138348103, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0020833334419876337, "reward_total_mean": 0.9976277351379395, "reward_meter_mean": 0.9976277351379395, "reward_meter_std": 4.747844286612235e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9976277351379395, "reward_total_composite_std": 4.747844286612235e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1302.0} {"timestamp_utc": "2026-04-11T22:11:15Z", "mode": "train", "global_step": 1303, "epoch": 0.05031665122026568, "loss": -0.0248, "grad_norm": 2.1486196517944336, "learning_rate": 6.0545454545454555e-06, "num_tokens": 2817248.0, "completions/mean_length": 194.625, "completions/min_length": 176.0, "completions/max_length": 207.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 194.625, "completions/min_terminated_length": 176.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.9984779357910156, "rewards/meter/std": 0.0009284228435717523, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9568568468093872, "rewards/total_composite/std": 0.0768078938126564, "reward": 0.9568568468093872, "reward_std": 0.07680792361497879, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010335750877857208, "sampling/sampling_logp_difference/max": 1.4263226985931396, "sampling/importance_sampling_ratio/min": 0.2401905655860901, "sampling/importance_sampling_ratio/mean": 1.00020170211792, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03241105214692652, "clip_ratio/low_mean": 0.002705548540689051, "clip_ratio/low_min": 0.002705548540689051, "clip_ratio/high_mean": 0.008829656231682748, "clip_ratio/high_max": 0.008829656231682748, "clip_ratio/region_mean": 0.011535204772371799, "reward_total_mean": 0.9568568468093872, "reward_meter_mean": 0.9984779357910156, "reward_meter_std": 0.0009284228435717523, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9568568468093872, "reward_total_composite_std": 0.0768078938126564, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1303.0} {"timestamp_utc": "2026-04-11T22:11:22Z", "mode": "train", "global_step": 1304, "epoch": 0.0503552672227371, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.051515151515152e-06, "num_tokens": 2820424.0, "completions/mean_length": 189.0, "completions/min_length": 189.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 189.0, "completions/min_terminated_length": 189.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.9931953549385071, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9931953549385071, "rewards/total_composite/std": 0.0, "reward": 0.9931953549385071, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00013056160241831094, "sampling/sampling_logp_difference/max": 0.00871575903147459, "sampling/importance_sampling_ratio/min": 0.9913221001625061, "sampling/importance_sampling_ratio/mean": 1.000101923942566, "sampling/importance_sampling_ratio/max": 1.0045995712280273, "entropy": 0.0010099293140228838, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9931953549385071, "reward_meter_mean": 0.9931953549385071, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9931953549385071, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1304.0} {"timestamp_utc": "2026-04-11T22:11:26Z", "mode": "train", "global_step": 1305, "epoch": 0.050393883225208526, "loss": 0.0156, "grad_norm": 5.147261142730713, "learning_rate": 6.048484848484849e-06, "num_tokens": 2821849.0, "completions/mean_length": 28.125, "completions/min_length": 28.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9834710359573364, "rewards/meter/std": 0.003339758375659585, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9834710359573364, "rewards/total_composite/std": 0.003339758375659585, "reward": 0.9834710359573364, "reward_std": 0.0033397525548934937, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018510645255446434, "sampling/sampling_logp_difference/max": 0.5660119652748108, "sampling/importance_sampling_ratio/min": 0.5677852630615234, "sampling/importance_sampling_ratio/mean": 0.9973562359809875, "sampling/importance_sampling_ratio/max": 1.480722188949585, "entropy": 0.06381317507475615, "clip_ratio/low_mean": 0.008620689623057842, "clip_ratio/low_min": 0.008620689623057842, "clip_ratio/high_mean": 0.01785714365541935, "clip_ratio/high_max": 0.01785714365541935, "clip_ratio/region_mean": 0.026477833278477192, "reward_total_mean": 0.9834710359573364, "reward_meter_mean": 0.9834710359573364, "reward_meter_std": 0.003339758375659585, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9834710359573364, "reward_total_composite_std": 0.003339758375659585, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1305.0} {"timestamp_utc": "2026-04-11T22:11:31Z", "mode": "train", "global_step": 1306, "epoch": 0.05043249922767995, "loss": -0.004, "grad_norm": 5.392800331115723, "learning_rate": 6.0454545454545456e-06, "num_tokens": 2823676.0, "completions/mean_length": 71.375, "completions/min_length": 66.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.1207331120967865, "rewards/meter/std": 0.06592007726430893, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.10389591753482819, "rewards/total_composite/std": 0.029868489131331444, "reward": 0.10389591753482819, "reward_std": 0.029868490993976593, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014333794824779034, "sampling/sampling_logp_difference/max": 0.7225782871246338, "sampling/importance_sampling_ratio/min": 0.48549890518188477, "sampling/importance_sampling_ratio/mean": 1.0015692710876465, "sampling/importance_sampling_ratio/max": 1.8720688819885254, "entropy": 0.05604141438379884, "clip_ratio/low_mean": 0.0017123287543654442, "clip_ratio/low_min": 0.0017123287543654442, "clip_ratio/high_mean": 0.006989758228883147, "clip_ratio/high_max": 0.006989758228883147, "clip_ratio/region_mean": 0.008702086983248591, "reward_total_mean": 0.10389591753482819, "reward_meter_mean": 0.1207331120967865, "reward_meter_std": 0.06592007726430893, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.10389591753482819, "reward_total_composite_std": 0.029868489131331444, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1306.0} {"timestamp_utc": "2026-04-11T22:11:36Z", "mode": "train", "global_step": 1307, "epoch": 0.050471115230151374, "loss": 0.1852, "grad_norm": 13.817164421081543, "learning_rate": 6.042424242424243e-06, "num_tokens": 2825336.0, "completions/mean_length": 50.5, "completions/min_length": 45.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.5, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.892030656337738, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7805268168449402, "rewards/total_composite/std": 0.2064649760723114, "reward": 0.7805268168449402, "reward_std": 0.2064649760723114, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014084003865718842, "sampling/sampling_logp_difference/max": 1.5264016389846802, "sampling/importance_sampling_ratio/min": 0.21731625497341156, "sampling/importance_sampling_ratio/mean": 1.0002713203430176, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0181692021433264, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.008333333535119891, "clip_ratio/high_max": 0.008333333535119891, "clip_ratio/region_mean": 0.010199005133472383, "reward_total_mean": 0.7805268168449402, "reward_meter_mean": 0.892030656337738, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7805268168449402, "reward_total_composite_std": 0.2064649760723114, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1307.0} {"timestamp_utc": "2026-04-11T22:11:46Z", "mode": "train", "global_step": 1308, "epoch": 0.0505097312326228, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.039393939393939e-06, "num_tokens": 2826688.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.0, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0, "rewards/total_composite/std": 0.0, "reward": 0.0, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.0, "reward_meter_mean": 0.0, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1308.0} {"timestamp_utc": "2026-04-11T22:11:53Z", "mode": "train", "global_step": 1309, "epoch": 0.05054834723509422, "loss": 0.0177, "grad_norm": 1.092119812965393, "learning_rate": 6.0363636363636365e-06, "num_tokens": 2829681.0, "completions/mean_length": 181.125, "completions/min_length": 177.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.125, "completions/min_terminated_length": 177.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.941912055015564, "rewards/meter/std": 0.057046957314014435, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.941912055015564, "rewards/total_composite/std": 0.057046957314014435, "reward": 0.941912055015564, "reward_std": 0.057046957314014435, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009103724732995033, "sampling/sampling_logp_difference/max": 1.9865416288375854, "sampling/importance_sampling_ratio/min": 0.1371689885854721, "sampling/importance_sampling_ratio/mean": 1.0001729726791382, "sampling/importance_sampling_ratio/max": 1.3597184419631958, "entropy": 0.03956991946324706, "clip_ratio/low_mean": 0.0026963776908814907, "clip_ratio/low_min": 0.0026963776908814907, "clip_ratio/high_mean": 0.00348425266565755, "clip_ratio/high_max": 0.00348425266565755, "clip_ratio/region_mean": 0.006180630356539041, "reward_total_mean": 0.941912055015564, "reward_meter_mean": 0.941912055015564, "reward_meter_std": 0.057046957314014435, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.941912055015564, "reward_total_composite_std": 0.057046957314014435, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1309.0} {"timestamp_utc": "2026-04-11T22:11:59Z", "mode": "train", "global_step": 1310, "epoch": 0.050586963237565646, "loss": 0.0373, "grad_norm": 5.895564079284668, "learning_rate": 6.033333333333335e-06, "num_tokens": 2832425.0, "completions/mean_length": 168.0, "completions/min_length": 162.0, "completions/max_length": 184.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.0, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 184.0, "rewards/meter/mean": 0.8207767009735107, "rewards/meter/std": 0.3043481707572937, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8207767009735107, "rewards/total_composite/std": 0.3043481707572937, "reward": 0.8207767009735107, "reward_std": 0.3043481707572937, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0176156684756279, "sampling/sampling_logp_difference/max": 1.3011794090270996, "sampling/importance_sampling_ratio/min": 0.27221056818962097, "sampling/importance_sampling_ratio/mean": 0.9986696839332581, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06413523061200976, "clip_ratio/low_mean": 0.0050322061870247126, "clip_ratio/low_min": 0.0050322061870247126, "clip_ratio/high_mean": 0.009749046759679914, "clip_ratio/high_max": 0.009749046759679914, "clip_ratio/region_mean": 0.014781252946704626, "reward_total_mean": 0.8207767009735107, "reward_meter_mean": 0.8207767009735107, "reward_meter_std": 0.3043481707572937, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8207767009735107, "reward_total_composite_std": 0.3043481707572937, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1310.0} {"timestamp_utc": "2026-04-11T22:12:03Z", "mode": "train", "global_step": 1311, "epoch": 0.05062557924003707, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.030303030303031e-06, "num_tokens": 2834033.0, "completions/mean_length": 52.0, "completions/min_length": 52.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8786622881889343, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8786622881889343, "rewards/total_composite/std": 0.0, "reward": 0.8786622881889343, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00039858516538515687, "sampling/sampling_logp_difference/max": 0.012473315931856632, "sampling/importance_sampling_ratio/min": 0.9986094236373901, "sampling/importance_sampling_ratio/mean": 1.0003914833068848, "sampling/importance_sampling_ratio/max": 1.0125514268875122, "entropy": 0.0028791853110305965, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8786622881889343, "reward_meter_mean": 0.8786622881889343, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8786622881889343, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1311.0} {"timestamp_utc": "2026-04-11T22:12:09Z", "mode": "train", "global_step": 1312, "epoch": 0.050664195242508495, "loss": -0.0166, "grad_norm": 2.024810314178467, "learning_rate": 6.027272727272728e-06, "num_tokens": 2836054.0, "completions/mean_length": 97.625, "completions/min_length": 91.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.625, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.3544048070907593, "rewards/meter/std": 0.2585018575191498, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3544048070907593, "rewards/total_composite/std": 0.2585018575191498, "reward": 0.3544048070907593, "reward_std": 0.2585018575191498, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009315414354205132, "sampling/sampling_logp_difference/max": 0.7790820598602295, "sampling/importance_sampling_ratio/min": 0.45882701873779297, "sampling/importance_sampling_ratio/mean": 1.003443956375122, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03785029170103371, "clip_ratio/low_mean": 0.006459464086219668, "clip_ratio/low_min": 0.006459464086219668, "clip_ratio/high_mean": 0.0012254902394488454, "clip_ratio/high_max": 0.0012254902394488454, "clip_ratio/region_mean": 0.007684954325668514, "reward_total_mean": 0.3544048070907593, "reward_meter_mean": 0.3544048070907593, "reward_meter_std": 0.2585018575191498, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3544048070907593, "reward_total_composite_std": 0.2585018575191498, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1312.0} {"timestamp_utc": "2026-04-11T22:12:19Z", "mode": "train", "global_step": 1313, "epoch": 0.05070281124497992, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.024242424242425e-06, "num_tokens": 2837270.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.0, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0, "rewards/total_composite/std": 0.0, "reward": 0.0, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.0, "reward_meter_mean": 0.0, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1313.0} {"timestamp_utc": "2026-04-11T22:12:29Z", "mode": "train", "global_step": 1314, "epoch": 0.05074142724745134, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.021212121212122e-06, "num_tokens": 2838502.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.0, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0, "rewards/total_composite/std": 0.0, "reward": 0.0, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.0, "reward_meter_mean": 0.0, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1314.0} {"timestamp_utc": "2026-04-11T22:12:34Z", "mode": "train", "global_step": 1315, "epoch": 0.05078004324992277, "loss": -0.003, "grad_norm": 11.59805679321289, "learning_rate": 6.018181818181818e-06, "num_tokens": 2840007.0, "completions/mean_length": 36.125, "completions/min_length": 36.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9978696703910828, "rewards/meter/std": 0.00023855116160120815, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9978696703910828, "rewards/total_composite/std": 0.00023855116160120815, "reward": 0.9978696703910828, "reward_std": 0.00023855116160120815, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010207581333816051, "sampling/sampling_logp_difference/max": 1.1928129196166992, "sampling/importance_sampling_ratio/min": 0.3033667206764221, "sampling/importance_sampling_ratio/mean": 0.9941632747650146, "sampling/importance_sampling_ratio/max": 1.0860706567764282, "entropy": 0.008218832779675722, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0034722222480922937, "reward_total_mean": 0.9978696703910828, "reward_meter_mean": 0.9978696703910828, "reward_meter_std": 0.00023855116160120815, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9978696703910828, "reward_total_composite_std": 0.00023855116160120815, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1315.0} {"timestamp_utc": "2026-04-11T22:12:43Z", "mode": "train", "global_step": 1316, "epoch": 0.05081865925239419, "loss": 0.0033, "grad_norm": 1.2825982570648193, "learning_rate": 6.015151515151516e-06, "num_tokens": 2845365.0, "completions/mean_length": 418.75, "completions/min_length": 415.0, "completions/max_length": 422.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 418.75, "completions/min_terminated_length": 415.0, "completions/max_terminated_length": 422.0, "rewards/meter/mean": 0.993027925491333, "rewards/meter/std": 0.0008707118104211986, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.993027925491333, "rewards/total_composite/std": 0.0008707118104211986, "reward": 0.993027925491333, "reward_std": 0.0008707197848707438, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0037374256644397974, "sampling/sampling_logp_difference/max": 2.490666627883911, "sampling/importance_sampling_ratio/min": 0.08285471051931381, "sampling/importance_sampling_ratio/mean": 0.9987327456474304, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.007259320729644969, "clip_ratio/low_mean": 0.0005924170836806297, "clip_ratio/low_min": 0.0005924170836806297, "clip_ratio/high_mean": 0.003288808075012639, "clip_ratio/high_max": 0.003288808075012639, "clip_ratio/region_mean": 0.003881225158693269, "reward_total_mean": 0.993027925491333, "reward_meter_mean": 0.993027925491333, "reward_meter_std": 0.0008707118104211986, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.993027925491333, "reward_total_composite_std": 0.0008707118104211986, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1316.0} {"timestamp_utc": "2026-04-11T22:12:48Z", "mode": "train", "global_step": 1317, "epoch": 0.050857275254865615, "loss": -0.0232, "grad_norm": 6.237240314483643, "learning_rate": 6.012121212121213e-06, "num_tokens": 2847010.0, "completions/mean_length": 56.625, "completions/min_length": 55.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.625, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.3956717848777771, "rewards/meter/std": 0.47075986862182617, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.3956717848777771, "rewards/total_composite/std": 0.47075986862182617, "reward": 0.3956717848777771, "reward_std": 0.47075986862182617, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05238794535398483, "sampling/sampling_logp_difference/max": 5.138589859008789, "sampling/importance_sampling_ratio/min": 0.005865955725312233, "sampling/importance_sampling_ratio/mean": 0.9871724843978882, "sampling/importance_sampling_ratio/max": 1.6266334056854248, "entropy": 0.03596246731467545, "clip_ratio/low_mean": 0.011204146547242999, "clip_ratio/low_min": 0.011204146547242999, "clip_ratio/high_mean": 0.006398809840902686, "clip_ratio/high_max": 0.006398809840902686, "clip_ratio/region_mean": 0.017602956388145685, "reward_total_mean": 0.3956717848777771, "reward_meter_mean": 0.3956717848777771, "reward_meter_std": 0.47075986862182617, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.3956717848777771, "reward_total_composite_std": 0.47075986862182617, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1317.0} {"timestamp_utc": "2026-04-11T22:12:55Z", "mode": "train", "global_step": 1318, "epoch": 0.05089589125733704, "loss": 0.0113, "grad_norm": 1.826935052871704, "learning_rate": 6.00909090909091e-06, "num_tokens": 2848933.0, "completions/mean_length": 81.375, "completions/min_length": 72.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.375, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9984121918678284, "rewards/meter/std": 9.128195233643055e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984121918678284, "rewards/total_composite/std": 9.128195233643055e-05, "reward": 0.9984121918678284, "reward_std": 9.127175144385546e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0382012277841568, "sampling/sampling_logp_difference/max": 6.138245105743408, "sampling/importance_sampling_ratio/min": 0.0021587086375802755, "sampling/importance_sampling_ratio/mean": 0.9949803948402405, "sampling/importance_sampling_ratio/max": 1.6746786832809448, "entropy": 0.017093129688873887, "clip_ratio/low_mean": 0.01106532383710146, "clip_ratio/low_min": 0.01106532383710146, "clip_ratio/high_mean": 0.005856990348547697, "clip_ratio/high_max": 0.005856990348547697, "clip_ratio/region_mean": 0.016922314185649157, "reward_total_mean": 0.9984121918678284, "reward_meter_mean": 0.9984121918678284, "reward_meter_std": 9.128195233643055e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984121918678284, "reward_total_composite_std": 9.128195233643055e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1318.0} {"timestamp_utc": "2026-04-11T22:13:06Z", "mode": "train", "global_step": 1319, "epoch": 0.05093450725980846, "loss": -0.0754, "grad_norm": 0.8794658780097961, "learning_rate": 6.0060606060606065e-06, "num_tokens": 2850459.0, "completions/mean_length": 294.75, "completions/min_length": 74.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.2691759467124939, "rewards/meter/std": 0.4170503318309784, "rewards/count_adherence/mean": 0.5625, "rewards/count_adherence/std": 0.4955156147480011, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.2641150951385498, "rewards/total_composite/std": 0.41989636421203613, "reward": 0.2641150951385498, "reward_std": 0.41989633440971375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0324825793504715, "sampling/sampling_logp_difference/max": 1.2901360988616943, "sampling/importance_sampling_ratio/min": 0.27523329854011536, "sampling/importance_sampling_ratio/mean": 1.000800609588623, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0588919089641422, "clip_ratio/low_mean": 0.007852057227864861, "clip_ratio/low_min": 0.007852057227864861, "clip_ratio/high_mean": 0.00844594556838274, "clip_ratio/high_max": 0.00844594556838274, "clip_ratio/region_mean": 0.0162980027962476, "reward_total_mean": 0.2641150951385498, "reward_meter_mean": 0.2691759467124939, "reward_meter_std": 0.4170503318309784, "reward_count_adherence_mean": 0.5625, "reward_count_adherence_std": 0.4955156147480011, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.2641150951385498, "reward_total_composite_std": 0.41989636421203613, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1319.0} {"timestamp_utc": "2026-04-11T22:13:11Z", "mode": "train", "global_step": 1320, "epoch": 0.05097312326227989, "loss": -0.0019, "grad_norm": 6.607697486877441, "learning_rate": 6.003030303030304e-06, "num_tokens": 2852226.0, "completions/mean_length": 60.875, "completions/min_length": 60.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9897077083587646, "rewards/meter/std": 0.0019390290835872293, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9897077083587646, "rewards/total_composite/std": 0.0019390290835872293, "reward": 0.9897077083587646, "reward_std": 0.001939031993970275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002292489632964134, "sampling/sampling_logp_difference/max": 0.08471393585205078, "sampling/importance_sampling_ratio/min": 0.9187750816345215, "sampling/importance_sampling_ratio/mean": 1.0013474225997925, "sampling/importance_sampling_ratio/max": 1.0323666334152222, "entropy": 0.021433729911223054, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9897077083587646, "reward_meter_mean": 0.9897077083587646, "reward_meter_std": 0.0019390290835872293, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9897077083587646, "reward_total_composite_std": 0.0019390290835872293, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1320.0} {"timestamp_utc": "2026-04-11T22:13:15Z", "mode": "train", "global_step": 1321, "epoch": 0.05101173926475131, "loss": 0.0074, "grad_norm": 9.826302528381348, "learning_rate": 6e-06, "num_tokens": 2853693.0, "completions/mean_length": 30.375, "completions/min_length": 30.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.375, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.985051155090332, "rewards/meter/std": 0.03771612420678139, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.985051155090332, "rewards/total_composite/std": 0.03771612420678139, "reward": 0.985051155090332, "reward_std": 0.03771614283323288, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02259841561317444, "sampling/sampling_logp_difference/max": 1.3937053680419922, "sampling/importance_sampling_ratio/min": 0.24815408885478973, "sampling/importance_sampling_ratio/mean": 0.9992597103118896, "sampling/importance_sampling_ratio/max": 1.37688410282135, "entropy": 0.08633835706859827, "clip_ratio/low_mean": 0.004166666883975267, "clip_ratio/low_min": 0.004166666883975267, "clip_ratio/high_mean": 0.007938507944345474, "clip_ratio/high_max": 0.007938507944345474, "clip_ratio/region_mean": 0.012105174828320742, "reward_total_mean": 0.985051155090332, "reward_meter_mean": 0.985051155090332, "reward_meter_std": 0.03771612420678139, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.985051155090332, "reward_total_composite_std": 0.03771612420678139, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1321.0} {"timestamp_utc": "2026-04-11T22:13:20Z", "mode": "train", "global_step": 1322, "epoch": 0.051050355267222736, "loss": 0.0056, "grad_norm": 6.936736583709717, "learning_rate": 5.996969696969697e-06, "num_tokens": 2855469.0, "completions/mean_length": 74.0, "completions/min_length": 71.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.0, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.6407962441444397, "rewards/meter/std": 0.401462584733963, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6407962441444397, "rewards/total_composite/std": 0.401462584733963, "reward": 0.6407962441444397, "reward_std": 0.4014625549316406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03448081016540527, "sampling/sampling_logp_difference/max": 2.137319564819336, "sampling/importance_sampling_ratio/min": 0.11797063052654266, "sampling/importance_sampling_ratio/mean": 0.9993025660514832, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1406972468830645, "clip_ratio/low_mean": 0.008566171862185001, "clip_ratio/low_min": 0.008566171862185001, "clip_ratio/high_mean": 0.01178362569771707, "clip_ratio/high_max": 0.01178362569771707, "clip_ratio/region_mean": 0.020349797559902072, "reward_total_mean": 0.6407962441444397, "reward_meter_mean": 0.6407962441444397, "reward_meter_std": 0.401462584733963, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6407962441444397, "reward_total_composite_std": 0.401462584733963, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1322.0} {"timestamp_utc": "2026-04-11T22:13:31Z", "mode": "train", "global_step": 1323, "epoch": 0.05108897126969416, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.993939393939394e-06, "num_tokens": 2857013.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9984541535377502, "rewards/meter/std": 1.664801311562769e-05, "rewards/count_adherence/mean": 0.699999988079071, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6989179253578186, "rewards/total_composite/std": 1.1653606634354219e-05, "reward": 0.6989179253578186, "reward_std": 1.1644577170955017e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.6989179253578186, "reward_meter_mean": 0.9984541535377502, "reward_meter_std": 1.664801311562769e-05, "reward_count_adherence_mean": 0.699999988079071, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6989179253578186, "reward_total_composite_std": 1.1653606634354219e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1323.0} {"timestamp_utc": "2026-04-11T22:13:36Z", "mode": "train", "global_step": 1324, "epoch": 0.051127587272165584, "loss": 0.0117, "grad_norm": 17.978578567504883, "learning_rate": 5.990909090909092e-06, "num_tokens": 2859469.0, "completions/mean_length": 120.0, "completions/min_length": 112.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.0, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.7927520275115967, "rewards/meter/std": 0.29552599787712097, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7927520275115967, "rewards/total_composite/std": 0.29552599787712097, "reward": 0.7927520275115967, "reward_std": 0.2955259680747986, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020901571959257126, "sampling/sampling_logp_difference/max": 3.8544344902038574, "sampling/importance_sampling_ratio/min": 0.021185578778386116, "sampling/importance_sampling_ratio/mean": 0.9972667098045349, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.057597613194957376, "clip_ratio/low_mean": 0.0019841270986944437, "clip_ratio/low_min": 0.0019841270986944437, "clip_ratio/high_mean": 0.008277167566120625, "clip_ratio/high_max": 0.008277167566120625, "clip_ratio/region_mean": 0.010261294664815068, "reward_total_mean": 0.7927520275115967, "reward_meter_mean": 0.7927520275115967, "reward_meter_std": 0.29552599787712097, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7927520275115967, "reward_total_composite_std": 0.29552599787712097, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1324.0} {"timestamp_utc": "2026-04-11T22:13:44Z", "mode": "train", "global_step": 1325, "epoch": 0.05116620327463701, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.987878787878788e-06, "num_tokens": 2863659.0, "completions/mean_length": 304.75, "completions/min_length": 289.0, "completions/max_length": 307.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 304.75, "completions/min_terminated_length": 289.0, "completions/max_terminated_length": 307.0, "rewards/meter/mean": 0.998460054397583, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.8888888955116272, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8875200748443604, "rewards/total_composite/std": 0.0, "reward": 0.8875200748443604, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0011180409928783774, "sampling/sampling_logp_difference/max": 0.7125938534736633, "sampling/importance_sampling_ratio/min": 0.49037057161331177, "sampling/importance_sampling_ratio/mean": 1.0002341270446777, "sampling/importance_sampling_ratio/max": 1.4610806703567505, "entropy": 0.002308362767507788, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8875200748443604, "reward_meter_mean": 0.998460054397583, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.8888888955116272, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8875200748443604, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1325.0} {"timestamp_utc": "2026-04-11T22:13:50Z", "mode": "train", "global_step": 1326, "epoch": 0.05120481927710843, "loss": -0.0069, "grad_norm": 1.9113426208496094, "learning_rate": 5.984848484848486e-06, "num_tokens": 2866046.0, "completions/mean_length": 125.375, "completions/min_length": 122.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.375, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.9981350898742676, "rewards/meter/std": 0.0005531536880880594, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9981350898742676, "rewards/total_composite/std": 0.0005531536880880594, "reward": 0.9981350898742676, "reward_std": 0.0005531637580133975, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019559325650334358, "sampling/sampling_logp_difference/max": 3.32448410987854, "sampling/importance_sampling_ratio/min": 0.03599108010530472, "sampling/importance_sampling_ratio/mean": 0.9965345859527588, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.019163366232533008, "clip_ratio/low_mean": 0.005098360474221408, "clip_ratio/low_min": 0.005098360474221408, "clip_ratio/high_mean": 0.007936508394777775, "clip_ratio/high_max": 0.007936508394777775, "clip_ratio/region_mean": 0.013034868868999183, "reward_total_mean": 0.9981350898742676, "reward_meter_mean": 0.9981350898742676, "reward_meter_std": 0.0005531536880880594, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9981350898742676, "reward_total_composite_std": 0.0005531536880880594, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1326.0} {"timestamp_utc": "2026-04-11T22:14:00Z", "mode": "train", "global_step": 1327, "epoch": 0.051243435279579856, "loss": -0.2053, "grad_norm": 0.9582489132881165, "learning_rate": 5.981818181818182e-06, "num_tokens": 2868176.0, "completions/mean_length": 445.25, "completions/min_length": 221.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.75, "completions/mean_terminated_length": 245.0, "completions/min_terminated_length": 221.0, "completions/max_terminated_length": 269.0, "rewards/meter/mean": 0.290924608707428, "rewards/meter/std": 0.41494548320770264, "rewards/count_adherence/mean": 0.2857142984867096, "rewards/count_adherence/std": 0.45175397396087646, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/total_composite/mean": 0.18730738759040833, "rewards/total_composite/std": 0.3614543676376343, "reward": 0.18730738759040833, "reward_std": 0.36145439743995667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01794980838894844, "sampling/sampling_logp_difference/max": 1.1533335447311401, "sampling/importance_sampling_ratio/min": 0.3155830204486847, "sampling/importance_sampling_ratio/mean": 1.0009292364120483, "sampling/importance_sampling_ratio/max": 1.7363885641098022, "entropy": 0.02190964389592409, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0035555686336010695, "clip_ratio/high_max": 0.0035555686336010695, "clip_ratio/region_mean": 0.0035555686336010695, "reward_total_mean": 0.18730738759040833, "reward_meter_mean": 0.290924608707428, "reward_meter_std": 0.41494548320770264, "reward_count_adherence_mean": 0.2857142984867096, "reward_count_adherence_std": 0.45175397396087646, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_total_composite_mean": 0.18730738759040833, "reward_total_composite_std": 0.3614543676376343, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1327.0} {"timestamp_utc": "2026-04-11T22:14:05Z", "mode": "train", "global_step": 1328, "epoch": 0.05128205128205128, "loss": 0.031, "grad_norm": 6.503256797790527, "learning_rate": 5.978787878787879e-06, "num_tokens": 2870086.0, "completions/mean_length": 70.75, "completions/min_length": 67.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.75, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.22972472012043, "rewards/meter/std": 0.1322513222694397, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.18738967180252075, "rewards/total_composite/std": 0.10936737805604935, "reward": 0.18738967180252075, "reward_std": 0.10936737805604935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10945596545934677, "sampling/sampling_logp_difference/max": 6.909653186798096, "sampling/importance_sampling_ratio/min": 0.0009981038747355342, "sampling/importance_sampling_ratio/mean": 0.9740753769874573, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06366277718916535, "clip_ratio/low_mean": 0.0033938727574422956, "clip_ratio/low_min": 0.0033938727574422956, "clip_ratio/high_mean": 0.02321748191025108, "clip_ratio/high_max": 0.02321748191025108, "clip_ratio/region_mean": 0.026611354667693377, "reward_total_mean": 0.18738967180252075, "reward_meter_mean": 0.22972472012043, "reward_meter_std": 0.1322513222694397, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.18738967180252075, "reward_total_composite_std": 0.10936737805604935, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1328.0} {"timestamp_utc": "2026-04-11T22:14:10Z", "mode": "train", "global_step": 1329, "epoch": 0.051320667284522704, "loss": 0.0334, "grad_norm": 3.808407783508301, "learning_rate": 5.975757575757576e-06, "num_tokens": 2871885.0, "completions/mean_length": 81.875, "completions/min_length": 70.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.875, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.9848395586013794, "rewards/meter/std": 0.0020317044109106064, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9848395586013794, "rewards/total_composite/std": 0.0020317044109106064, "reward": 0.9848395586013794, "reward_std": 0.002031713956966996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018270742148160934, "sampling/sampling_logp_difference/max": 1.6047842502593994, "sampling/importance_sampling_ratio/min": 0.20093289017677307, "sampling/importance_sampling_ratio/mean": 1.0018320083618164, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.042814851738512516, "clip_ratio/low_mean": 0.009586586384102702, "clip_ratio/low_min": 0.009586586384102702, "clip_ratio/high_mean": 0.008350278600119054, "clip_ratio/high_max": 0.008350278600119054, "clip_ratio/region_mean": 0.017936864984221756, "reward_total_mean": 0.9848395586013794, "reward_meter_mean": 0.9848395586013794, "reward_meter_std": 0.0020317044109106064, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9848395586013794, "reward_total_composite_std": 0.0020317044109106064, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1329.0} {"timestamp_utc": "2026-04-11T22:14:18Z", "mode": "train", "global_step": 1330, "epoch": 0.05135928328699413, "loss": -0.029, "grad_norm": 1.7039738893508911, "learning_rate": 5.972727272727274e-06, "num_tokens": 2876006.0, "completions/mean_length": 313.125, "completions/min_length": 296.0, "completions/max_length": 323.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 313.125, "completions/min_terminated_length": 296.0, "completions/max_terminated_length": 323.0, "rewards/meter/mean": 0.9908891916275024, "rewards/meter/std": 0.0022803200408816338, "rewards/count_adherence/mean": 0.9090909361839294, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9008083343505859, "rewards/total_composite/std": 0.002073025330901146, "reward": 0.9008083343505859, "reward_std": 0.0020730202086269855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006615650374442339, "sampling/sampling_logp_difference/max": 1.0105345249176025, "sampling/importance_sampling_ratio/min": 0.3640243709087372, "sampling/importance_sampling_ratio/mean": 0.9992187023162842, "sampling/importance_sampling_ratio/max": 1.394850730895996, "entropy": 0.02753951121121645, "clip_ratio/low_mean": 0.0008375796896871179, "clip_ratio/low_min": 0.0008375796896871179, "clip_ratio/high_mean": 0.0027186230290681124, "clip_ratio/high_max": 0.0027186230290681124, "clip_ratio/region_mean": 0.0035562027187552303, "reward_total_mean": 0.9008083343505859, "reward_meter_mean": 0.9908891916275024, "reward_meter_std": 0.0022803200408816338, "reward_count_adherence_mean": 0.9090909361839294, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9008083343505859, "reward_total_composite_std": 0.002073025330901146, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1330.0} {"timestamp_utc": "2026-04-11T22:14:23Z", "mode": "train", "global_step": 1331, "epoch": 0.05139789928946555, "loss": 0.0789, "grad_norm": 7.061683177947998, "learning_rate": 5.96969696969697e-06, "num_tokens": 2877932.0, "completions/mean_length": 79.75, "completions/min_length": 76.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.75, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8737680912017822, "rewards/total_composite/std": 0.23112915456295013, "reward": 0.8737680912017822, "reward_std": 0.23112915456295013, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007187447044998407, "sampling/sampling_logp_difference/max": 2.0118331909179688, "sampling/importance_sampling_ratio/min": 0.1337432712316513, "sampling/importance_sampling_ratio/mean": 1.0011377334594727, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.008815839100861922, "clip_ratio/low_mean": 0.0013736264081671834, "clip_ratio/low_min": 0.0013736264081671834, "clip_ratio/high_mean": 0.0016447368543595076, "clip_ratio/high_max": 0.0016447368543595076, "clip_ratio/region_mean": 0.003018363262526691, "reward_total_mean": 0.8737680912017822, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8737680912017822, "reward_total_composite_std": 0.23112915456295013, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1331.0} {"timestamp_utc": "2026-04-11T22:14:33Z", "mode": "train", "global_step": 1332, "epoch": 0.05143651529193698, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.966666666666667e-06, "num_tokens": 2879164.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.0, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0, "rewards/total_composite/std": 0.0, "reward": 0.0, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.0, "reward_meter_mean": 0.0, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1332.0} {"timestamp_utc": "2026-04-11T22:14:39Z", "mode": "train", "global_step": 1333, "epoch": 0.0514751312944084, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.963636363636364e-06, "num_tokens": 2880916.0, "completions/mean_length": 52.0, "completions/min_length": 52.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8786622881889343, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8786622881889343, "rewards/total_composite/std": 0.0, "reward": 0.8786622881889343, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0004812732804566622, "sampling/sampling_logp_difference/max": 0.025802575051784515, "sampling/importance_sampling_ratio/min": 0.9745274782180786, "sampling/importance_sampling_ratio/mean": 1.0001213550567627, "sampling/importance_sampling_ratio/max": 1.0151917934417725, "entropy": 0.0043671890161931515, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8786622881889343, "reward_meter_mean": 0.8786622881889343, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8786622881889343, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1333.0} {"timestamp_utc": "2026-04-11T22:14:47Z", "mode": "train", "global_step": 1334, "epoch": 0.051513747296879825, "loss": 0.0026, "grad_norm": 0.39917171001434326, "learning_rate": 5.960606060606061e-06, "num_tokens": 2884680.0, "completions/mean_length": 255.5, "completions/min_length": 243.0, "completions/max_length": 259.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 255.5, "completions/min_terminated_length": 243.0, "completions/max_terminated_length": 259.0, "rewards/meter/mean": 0.004360870458185673, "rewards/meter/std": 0.00025717844255268574, "rewards/count_adherence/mean": 0.9722222089767456, "rewards/count_adherence/std": 0.05143444985151291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.004241206683218479, "rewards/total_composite/std": 0.0003409688069950789, "reward": 0.004241206683218479, "reward_std": 0.0003409688069950789, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00860567856580019, "sampling/sampling_logp_difference/max": 4.417545318603516, "sampling/importance_sampling_ratio/min": 0.012063808739185333, "sampling/importance_sampling_ratio/mean": 0.9992878437042236, "sampling/importance_sampling_ratio/max": 1.939797043800354, "entropy": 0.01055589783936739, "clip_ratio/low_mean": 0.0004826254735235125, "clip_ratio/low_min": 0.0004826254735235125, "clip_ratio/high_mean": 0.0019455252913758159, "clip_ratio/high_max": 0.0019455252913758159, "clip_ratio/region_mean": 0.0024281507648993284, "reward_total_mean": 0.004241206683218479, "reward_meter_mean": 0.004360870458185673, "reward_meter_std": 0.00025717844255268574, "reward_count_adherence_mean": 0.9722222089767456, "reward_count_adherence_std": 0.05143444985151291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.004241206683218479, "reward_total_composite_std": 0.0003409688069950789, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1334.0} {"timestamp_utc": "2026-04-11T22:14:54Z", "mode": "train", "global_step": 1335, "epoch": 0.05155236329935125, "loss": 0.0234, "grad_norm": 2.817392587661743, "learning_rate": 5.9575757575757575e-06, "num_tokens": 2887026.0, "completions/mean_length": 129.25, "completions/min_length": 126.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.25, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.9876137971878052, "rewards/meter/std": 0.008682074956595898, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9876137971878052, "rewards/total_composite/std": 0.008682074956595898, "reward": 0.9876137971878052, "reward_std": 0.008682073093950748, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014027805998921394, "sampling/sampling_logp_difference/max": 1.8226474523544312, "sampling/importance_sampling_ratio/min": 0.1615973711013794, "sampling/importance_sampling_ratio/mean": 0.9986028075218201, "sampling/importance_sampling_ratio/max": 1.8998475074768066, "entropy": 0.0517108803614974, "clip_ratio/low_mean": 0.0038037330959923565, "clip_ratio/low_min": 0.0038037330959923565, "clip_ratio/high_mean": 0.005829803936649114, "clip_ratio/high_max": 0.005829803936649114, "clip_ratio/region_mean": 0.00963353703264147, "reward_total_mean": 0.9876137971878052, "reward_meter_mean": 0.9876137971878052, "reward_meter_std": 0.008682074956595898, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9876137971878052, "reward_total_composite_std": 0.008682074956595898, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1335.0} {"timestamp_utc": "2026-04-11T22:14:59Z", "mode": "train", "global_step": 1336, "epoch": 0.05159097930182267, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.954545454545455e-06, "num_tokens": 2889122.0, "completions/mean_length": 106.0, "completions/min_length": 106.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.0, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00049329933244735, "sampling/sampling_logp_difference/max": 0.058125704526901245, "sampling/importance_sampling_ratio/min": 0.9435313940048218, "sampling/importance_sampling_ratio/mean": 1.000313639640808, "sampling/importance_sampling_ratio/max": 1.0438523292541504, "entropy": 0.0051293845172040164, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1336.0} {"timestamp_utc": "2026-04-11T22:15:04Z", "mode": "train", "global_step": 1337, "epoch": 0.0516295953042941, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.951515151515151e-06, "num_tokens": 2890746.0, "completions/mean_length": 52.0, "completions/min_length": 52.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8786622881889343, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8786622881889343, "rewards/total_composite/std": 0.0, "reward": 0.8786622881889343, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0008134738891385496, "sampling/sampling_logp_difference/max": 0.08420295268297195, "sampling/importance_sampling_ratio/min": 0.9192447066307068, "sampling/importance_sampling_ratio/mean": 1.0001220703125, "sampling/importance_sampling_ratio/max": 1.0231122970581055, "entropy": 0.003981474466854706, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8786622881889343, "reward_meter_mean": 0.8786622881889343, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8786622881889343, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1337.0} {"timestamp_utc": "2026-04-11T22:15:10Z", "mode": "train", "global_step": 1338, "epoch": 0.05166821130676552, "loss": 0.0649, "grad_norm": 1.407727599143982, "learning_rate": 5.948484848484849e-06, "num_tokens": 2893747.0, "completions/mean_length": 182.125, "completions/min_length": 175.0, "completions/max_length": 216.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 182.125, "completions/min_terminated_length": 175.0, "completions/max_terminated_length": 216.0, "rewards/meter/mean": 0.9925892353057861, "rewards/meter/std": 0.0009907754138112068, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9718595147132874, "rewards/total_composite/std": 0.057647883892059326, "reward": 0.9718595147132874, "reward_std": 0.05764787644147873, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005145379342138767, "sampling/sampling_logp_difference/max": 1.0816781520843506, "sampling/importance_sampling_ratio/min": 0.3390261232852936, "sampling/importance_sampling_ratio/mean": 1.0014400482177734, "sampling/importance_sampling_ratio/max": 1.6286849975585938, "entropy": 0.024506430374458432, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/high_mean": 0.0028370333020575345, "clip_ratio/high_max": 0.0028370333020575345, "clip_ratio/region_mean": 0.004573144426103681, "reward_total_mean": 0.9718595147132874, "reward_meter_mean": 0.9925892353057861, "reward_meter_std": 0.0009907754138112068, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9718595147132874, "reward_total_composite_std": 0.057647883892059326, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1338.0} {"timestamp_utc": "2026-04-11T22:15:19Z", "mode": "train", "global_step": 1339, "epoch": 0.051706827309236945, "loss": 0.0477, "grad_norm": 8.732027053833008, "learning_rate": 5.9454545454545465e-06, "num_tokens": 2897894.0, "completions/mean_length": 283.375, "completions/min_length": 239.0, "completions/max_length": 331.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 283.375, "completions/min_terminated_length": 239.0, "completions/max_terminated_length": 331.0, "rewards/meter/mean": 0.9828551411628723, "rewards/meter/std": 0.03965521603822708, "rewards/count_adherence/mean": 0.9444444179534912, "rewards/count_adherence/std": 0.08399210125207901, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.927351176738739, "rewards/total_composite/std": 0.08097497373819351, "reward": 0.927351176738739, "reward_std": 0.08097497373819351, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009965971112251282, "sampling/sampling_logp_difference/max": 3.3419697284698486, "sampling/importance_sampling_ratio/min": 0.035367224365472794, "sampling/importance_sampling_ratio/mean": 0.9984091520309448, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0335227339528501, "clip_ratio/low_mean": 0.0025711811904329807, "clip_ratio/low_min": 0.0025711811904329807, "clip_ratio/high_mean": 0.006506105128210038, "clip_ratio/high_max": 0.006506105128210038, "clip_ratio/region_mean": 0.009077286318643019, "reward_total_mean": 0.927351176738739, "reward_meter_mean": 0.9828551411628723, "reward_meter_std": 0.03965521603822708, "reward_count_adherence_mean": 0.9444444179534912, "reward_count_adherence_std": 0.08399210125207901, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.927351176738739, "reward_total_composite_std": 0.08097497373819351, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1339.0} {"timestamp_utc": "2026-04-11T22:15:27Z", "mode": "train", "global_step": 1340, "epoch": 0.05174544331170837, "loss": 0.0085, "grad_norm": 1.1558176279067993, "learning_rate": 5.942424242424243e-06, "num_tokens": 2902384.0, "completions/mean_length": 339.25, "completions/min_length": 327.0, "completions/max_length": 341.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 339.25, "completions/min_terminated_length": 327.0, "completions/max_terminated_length": 341.0, "rewards/meter/mean": 0.0037263138219714165, "rewards/meter/std": 6.417154509108514e-05, "rewards/count_adherence/mean": 0.9204546213150024, "rewards/count_adherence/std": 0.03214120864868164, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0034316868986934423, "rewards/total_composite/std": 0.00018270131840836257, "reward": 0.0034316868986934423, "reward_std": 0.00018270131840836257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0019070998532697558, "sampling/sampling_logp_difference/max": 1.3373016119003296, "sampling/importance_sampling_ratio/min": 0.2625531852245331, "sampling/importance_sampling_ratio/mean": 1.000539779663086, "sampling/importance_sampling_ratio/max": 1.414039969444275, "entropy": 0.007557271514087915, "clip_ratio/low_mean": 0.0014662756584584713, "clip_ratio/low_min": 0.0014662756584584713, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0014662756584584713, "reward_total_mean": 0.0034316868986934423, "reward_meter_mean": 0.0037263138219714165, "reward_meter_std": 6.417154509108514e-05, "reward_count_adherence_mean": 0.9204546213150024, "reward_count_adherence_std": 0.03214120864868164, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0034316868986934423, "reward_total_composite_std": 0.00018270131840836257, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1340.0} {"timestamp_utc": "2026-04-11T22:15:31Z", "mode": "train", "global_step": 1341, "epoch": 0.051784059314179794, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.93939393939394e-06, "num_tokens": 2903800.0, "completions/mean_length": 34.0, "completions/min_length": 34.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.998616099357605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.998616099357605, "rewards/total_composite/std": 0.0, "reward": 0.998616099357605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.006542143411934376, "sampling/sampling_logp_difference/max": 0.1799483448266983, "sampling/importance_sampling_ratio/min": 0.9130212068557739, "sampling/importance_sampling_ratio/mean": 1.0039291381835938, "sampling/importance_sampling_ratio/max": 1.197155475616455, "entropy": 0.06870117178186774, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.998616099357605, "reward_meter_mean": 0.998616099357605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.998616099357605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1341.0} {"timestamp_utc": "2026-04-11T22:15:38Z", "mode": "train", "global_step": 1342, "epoch": 0.05182267531665122, "loss": 0.0708, "grad_norm": 2.8088293075561523, "learning_rate": 5.936363636363637e-06, "num_tokens": 2906611.0, "completions/mean_length": 169.375, "completions/min_length": 146.0, "completions/max_length": 202.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 169.375, "completions/min_terminated_length": 146.0, "completions/max_terminated_length": 202.0, "rewards/meter/mean": 0.992696225643158, "rewards/meter/std": 0.015125798992812634, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9688113927841187, "rewards/total_composite/std": 0.0826389268040657, "reward": 0.9688113927841187, "reward_std": 0.08263891935348511, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016292046755552292, "sampling/sampling_logp_difference/max": 3.1323540210723877, "sampling/importance_sampling_ratio/min": 0.043615005910396576, "sampling/importance_sampling_ratio/mean": 0.9987751245498657, "sampling/importance_sampling_ratio/max": 1.7292990684509277, "entropy": 0.07218409376218915, "clip_ratio/low_mean": 0.004331683274358511, "clip_ratio/low_min": 0.004331683274358511, "clip_ratio/high_mean": 0.011005136591847986, "clip_ratio/high_max": 0.011005136591847986, "clip_ratio/region_mean": 0.015336819866206497, "reward_total_mean": 0.9688113927841187, "reward_meter_mean": 0.992696225643158, "reward_meter_std": 0.015125798992812634, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9688113927841187, "reward_total_composite_std": 0.0826389268040657, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1342.0} {"timestamp_utc": "2026-04-11T22:15:46Z", "mode": "train", "global_step": 1343, "epoch": 0.05186129131912264, "loss": 0.0181, "grad_norm": 0.8513466715812683, "learning_rate": 5.933333333333335e-06, "num_tokens": 2911093.0, "completions/mean_length": 331.25, "completions/min_length": 283.0, "completions/max_length": 358.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 331.25, "completions/min_terminated_length": 283.0, "completions/max_terminated_length": 358.0, "rewards/meter/mean": 0.9972703456878662, "rewards/meter/std": 0.002285771304741502, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9972703456878662, "rewards/total_composite/std": 0.002285771304741502, "reward": 0.9972703456878662, "reward_std": 0.0022857647854834795, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009786021895706654, "sampling/sampling_logp_difference/max": 1.3884074687957764, "sampling/importance_sampling_ratio/min": 0.24947229027748108, "sampling/importance_sampling_ratio/mean": 1.0010948181152344, "sampling/importance_sampling_ratio/max": 1.9991581439971924, "entropy": 0.04378740070387721, "clip_ratio/low_mean": 0.001396648003719747, "clip_ratio/low_min": 0.001396648003719747, "clip_ratio/high_mean": 0.007295190705917776, "clip_ratio/high_max": 0.007295190705917776, "clip_ratio/region_mean": 0.008691838709637523, "reward_total_mean": 0.9972703456878662, "reward_meter_mean": 0.9972703456878662, "reward_meter_std": 0.002285771304741502, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9972703456878662, "reward_total_composite_std": 0.002285771304741502, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1343.0} {"timestamp_utc": "2026-04-11T22:15:51Z", "mode": "train", "global_step": 1344, "epoch": 0.051899907321594066, "loss": -0.0127, "grad_norm": 8.242257118225098, "learning_rate": 5.93030303030303e-06, "num_tokens": 2912935.0, "completions/mean_length": 64.25, "completions/min_length": 62.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8864076733589172, "rewards/meter/std": 0.31732821464538574, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8864076733589172, "rewards/total_composite/std": 0.31732821464538574, "reward": 0.8864076733589172, "reward_std": 0.31732818484306335, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021848265081644058, "sampling/sampling_logp_difference/max": 1.9131028652191162, "sampling/importance_sampling_ratio/min": 0.14762163162231445, "sampling/importance_sampling_ratio/mean": 0.9961261749267578, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06875600991770625, "clip_ratio/low_mean": 0.006048386916518211, "clip_ratio/low_min": 0.006048386916518211, "clip_ratio/high_mean": 0.019361895858310163, "clip_ratio/high_max": 0.019361895858310163, "clip_ratio/region_mean": 0.025410282774828374, "reward_total_mean": 0.8864076733589172, "reward_meter_mean": 0.8864076733589172, "reward_meter_std": 0.31732821464538574, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8864076733589172, "reward_total_composite_std": 0.31732821464538574, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1344.0} {"timestamp_utc": "2026-04-11T22:15:56Z", "mode": "train", "global_step": 1345, "epoch": 0.05193852332406549, "loss": -0.0063, "grad_norm": 1.4179552793502808, "learning_rate": 5.927272727272728e-06, "num_tokens": 2914823.0, "completions/mean_length": 69.0, "completions/min_length": 68.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9537819623947144, "rewards/meter/std": 0.0036520822905004025, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9537819623947144, "rewards/total_composite/std": 0.0036520822905004025, "reward": 0.9537819623947144, "reward_std": 0.003652075305581093, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0048215980641543865, "sampling/sampling_logp_difference/max": 0.8050665855407715, "sampling/importance_sampling_ratio/min": 0.44705820083618164, "sampling/importance_sampling_ratio/mean": 1.000336766242981, "sampling/importance_sampling_ratio/max": 1.0674445629119873, "entropy": 0.023285279516130686, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9537819623947144, "reward_meter_mean": 0.9537819623947144, "reward_meter_std": 0.0036520822905004025, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9537819623947144, "reward_total_composite_std": 0.0036520822905004025, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1345.0} {"timestamp_utc": "2026-04-11T22:16:01Z", "mode": "train", "global_step": 1346, "epoch": 0.051977139326536914, "loss": -0.0008, "grad_norm": 3.9453160762786865, "learning_rate": 5.924242424242425e-06, "num_tokens": 2916562.0, "completions/mean_length": 57.375, "completions/min_length": 57.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.375, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9918615221977234, "rewards/meter/std": 0.0014814144233241677, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9918615221977234, "rewards/total_composite/std": 0.0014814144233241677, "reward": 0.9918615221977234, "reward_std": 0.0014814181486144662, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017751315608620644, "sampling/sampling_logp_difference/max": 1.1306664943695068, "sampling/importance_sampling_ratio/min": 0.3228180408477783, "sampling/importance_sampling_ratio/mean": 0.9981111884117126, "sampling/importance_sampling_ratio/max": 1.563149094581604, "entropy": 0.06797249848023057, "clip_ratio/low_mean": 0.0021929824724793434, "clip_ratio/low_min": 0.0021929824724793434, "clip_ratio/high_mean": 0.012968844501301646, "clip_ratio/high_max": 0.012968844501301646, "clip_ratio/region_mean": 0.01516182697378099, "reward_total_mean": 0.9918615221977234, "reward_meter_mean": 0.9918615221977234, "reward_meter_std": 0.0014814144233241677, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9918615221977234, "reward_total_composite_std": 0.0014814144233241677, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1346.0} {"timestamp_utc": "2026-04-11T22:16:05Z", "mode": "train", "global_step": 1347, "epoch": 0.05201575532900834, "loss": -0.0089, "grad_norm": 4.1037397384643555, "learning_rate": 5.921212121212122e-06, "num_tokens": 2918264.0, "completions/mean_length": 56.75, "completions/min_length": 55.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.75, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9921625852584839, "rewards/meter/std": 0.0006445517647080123, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9921625852584839, "rewards/total_composite/std": 0.0006445517647080123, "reward": 0.9921625852584839, "reward_std": 0.0006445577600970864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004431902430951595, "sampling/sampling_logp_difference/max": 0.8569707870483398, "sampling/importance_sampling_ratio/min": 0.42444586753845215, "sampling/importance_sampling_ratio/mean": 0.9998234510421753, "sampling/importance_sampling_ratio/max": 1.0780442953109741, "entropy": 0.023745239013805985, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9921625852584839, "reward_meter_mean": 0.9921625852584839, "reward_meter_std": 0.0006445517647080123, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9921625852584839, "reward_total_composite_std": 0.0006445517647080123, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1347.0} {"timestamp_utc": "2026-04-11T22:16:10Z", "mode": "train", "global_step": 1348, "epoch": 0.05205437133147976, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.9181818181818184e-06, "num_tokens": 2920280.0, "completions/mean_length": 69.0, "completions/min_length": 69.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9552905559539795, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9552905559539795, "rewards/total_composite/std": 0.0, "reward": 0.9552905559539795, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0013839538441970944, "sampling/sampling_logp_difference/max": 0.025405190885066986, "sampling/importance_sampling_ratio/min": 0.9965260624885559, "sampling/importance_sampling_ratio/mean": 1.0013396739959717, "sampling/importance_sampling_ratio/max": 1.0257306098937988, "entropy": 0.013452943996526301, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9552905559539795, "reward_meter_mean": 0.9552905559539795, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9552905559539795, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1348.0} {"timestamp_utc": "2026-04-11T22:16:15Z", "mode": "train", "global_step": 1349, "epoch": 0.052092987333951186, "loss": -0.014, "grad_norm": 13.299444198608398, "learning_rate": 5.915151515151516e-06, "num_tokens": 2921925.0, "completions/mean_length": 49.625, "completions/min_length": 26.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.625, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.08599764108657837, "rewards/meter/std": 0.12936556339263916, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.08459585905075073, "rewards/total_composite/std": 0.1302107721567154, "reward": 0.08459585905075073, "reward_std": 0.1302107721567154, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06395962089300156, "sampling/sampling_logp_difference/max": 2.5426883697509766, "sampling/importance_sampling_ratio/min": 0.07865466177463531, "sampling/importance_sampling_ratio/mean": 1.0019546747207642, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1426252881065011, "clip_ratio/low_mean": 0.04252162668853998, "clip_ratio/low_min": 0.04252162668853998, "clip_ratio/high_mean": 0.0049019609577953815, "clip_ratio/high_max": 0.0049019609577953815, "clip_ratio/region_mean": 0.04742358764633536, "reward_total_mean": 0.08459585905075073, "reward_meter_mean": 0.08599764108657837, "reward_meter_std": 0.12936556339263916, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.08459585905075073, "reward_total_composite_std": 0.1302107721567154, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1349.0} {"timestamp_utc": "2026-04-11T22:16:20Z", "mode": "train", "global_step": 1350, "epoch": 0.05213160333642261, "loss": -0.0005, "grad_norm": 1.679924488067627, "learning_rate": 5.912121212121212e-06, "num_tokens": 2923883.0, "completions/mean_length": 76.75, "completions/min_length": 75.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.75, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9936152696609497, "rewards/meter/std": 0.00025513392756693065, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9936152696609497, "rewards/total_composite/std": 0.00025513392756693065, "reward": 0.9936152696609497, "reward_std": 0.0002551238867454231, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008142676204442978, "sampling/sampling_logp_difference/max": 1.6708736419677734, "sampling/importance_sampling_ratio/min": 0.18808266520500183, "sampling/importance_sampling_ratio/mean": 0.9977453351020813, "sampling/importance_sampling_ratio/max": 1.0962224006652832, "entropy": 0.02139825513586402, "clip_ratio/low_mean": 0.003246753243729472, "clip_ratio/low_min": 0.003246753243729472, "clip_ratio/high_mean": 0.0016666667070239782, "clip_ratio/high_max": 0.0016666667070239782, "clip_ratio/region_mean": 0.00491341995075345, "reward_total_mean": 0.9936152696609497, "reward_meter_mean": 0.9936152696609497, "reward_meter_std": 0.00025513392756693065, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9936152696609497, "reward_total_composite_std": 0.00025513392756693065, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1350.0} {"timestamp_utc": "2026-04-11T22:17:45Z", "mode": "eval", "global_step": 1350, "epoch": 0.05213160333642261, "eval_loss": NaN, "eval_runtime": 85.4628, "eval_samples_per_second": 1.217, "eval_steps_per_second": 0.152, "eval_num_tokens": 2923883.0, "eval_completions/mean_length": 218.21153846153845, "eval_completions/min_length": 50.15384615384615, "eval_completions/max_length": 459.0769230769231, "eval_completions/clipped_ratio": 0.09615384615384616, "eval_completions/mean_terminated_length": 187.16310002253607, "eval_completions/min_terminated_length": 50.15384615384615, "eval_completions/max_terminated_length": 362.2307692307692, "eval_rewards/meter/mean": 0.5815585691195267, "eval_rewards/meter/std": 0.4463339539674612, "eval_rewards/count_adherence/mean": 0.8320288841540997, "eval_rewards/count_adherence/std": 0.24656815941517168, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/total_composite/mean": 0.5448528723074839, "eval_rewards/total_composite/std": 0.4263327855330247, "eval_reward": 0.5448528723074839, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.003096834604198543, "eval_sampling/sampling_logp_difference/max": 0.675746697645921, "eval_sampling/importance_sampling_ratio/min": 0.5200849083753732, "eval_sampling/importance_sampling_ratio/mean": 1.0003560047883253, "eval_sampling/importance_sampling_ratio/max": 1.2023427119621863, "eval_entropy": 0.020505401019293528, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5448528723074839, "eval_reward_meter_mean": 0.5815585691195267, "eval_reward_meter_std": 0.4463339539674612, "eval_reward_count_adherence_mean": 0.8320288841540997, "eval_reward_count_adherence_std": 0.24656815941517168, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_total_composite_mean": 0.5448528723074839, "eval_reward_total_composite_std": 0.4263327855330247, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1350.0} {"timestamp_utc": "2026-04-11T22:17:54Z", "mode": "train", "global_step": 1351, "epoch": 0.052170219338894035, "loss": -0.002, "grad_norm": 0.1345411092042923, "learning_rate": 5.90909090909091e-06, "num_tokens": 2925286.0, "completions/mean_length": 31.375, "completions/min_length": 31.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.998595118522644, "rewards/meter/std": 8.492683264194056e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.998595118522644, "rewards/total_composite/std": 8.492683264194056e-06, "reward": 0.998595118522644, "reward_std": 8.489590072713327e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002988222986459732, "sampling/sampling_logp_difference/max": 0.11739683151245117, "sampling/importance_sampling_ratio/min": 0.8892322182655334, "sampling/importance_sampling_ratio/mean": 1.0004863739013672, "sampling/importance_sampling_ratio/max": 1.0749962329864502, "entropy": 0.021150008193217218, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0036764706019312143, "clip_ratio/high_max": 0.0036764706019312143, "clip_ratio/region_mean": 0.0036764706019312143, "reward_total_mean": 0.998595118522644, "reward_meter_mean": 0.998595118522644, "reward_meter_std": 8.492683264194056e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.998595118522644, "reward_total_composite_std": 8.492683264194056e-06, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1351.0} {"timestamp_utc": "2026-04-11T22:18:00Z", "mode": "train", "global_step": 1352, "epoch": 0.05220883534136546, "loss": -0.0169, "grad_norm": 3.95935320854187, "learning_rate": 5.906060606060607e-06, "num_tokens": 2927626.0, "completions/mean_length": 112.5, "completions/min_length": 107.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.5, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9887250661849976, "rewards/meter/std": 0.010928617790341377, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9887250661849976, "rewards/total_composite/std": 0.010928617790341377, "reward": 0.9887250661849976, "reward_std": 0.010928621515631676, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018162552267313004, "sampling/sampling_logp_difference/max": 4.197412014007568, "sampling/importance_sampling_ratio/min": 0.015034436248242855, "sampling/importance_sampling_ratio/mean": 0.994788646697998, "sampling/importance_sampling_ratio/max": 1.4109289646148682, "entropy": 0.024838423123583198, "clip_ratio/low_mean": 0.001168224262073636, "clip_ratio/low_min": 0.001168224262073636, "clip_ratio/high_mean": 0.003301642369478941, "clip_ratio/high_max": 0.003301642369478941, "clip_ratio/region_mean": 0.004469866631552577, "reward_total_mean": 0.9887250661849976, "reward_meter_mean": 0.9887250661849976, "reward_meter_std": 0.010928617790341377, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9887250661849976, "reward_total_composite_std": 0.010928617790341377, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1352.0} {"timestamp_utc": "2026-04-11T22:18:05Z", "mode": "train", "global_step": 1353, "epoch": 0.05224745134383688, "loss": 0.0236, "grad_norm": 17.220672607421875, "learning_rate": 5.903030303030304e-06, "num_tokens": 2929303.0, "completions/mean_length": 51.625, "completions/min_length": 49.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.625, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9894022941589355, "rewards/meter/std": 0.007141584530472755, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9894022941589355, "rewards/total_composite/std": 0.007141584530472755, "reward": 0.9894022941589355, "reward_std": 0.007141584530472755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04331376403570175, "sampling/sampling_logp_difference/max": 3.3641629219055176, "sampling/importance_sampling_ratio/min": 0.034590959548950195, "sampling/importance_sampling_ratio/mean": 0.9987140893936157, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06690713693387806, "clip_ratio/low_mean": 0.009433962404727936, "clip_ratio/low_min": 0.009433962404727936, "clip_ratio/high_mean": 0.01963563240133226, "clip_ratio/high_max": 0.01963563240133226, "clip_ratio/region_mean": 0.029069594806060195, "reward_total_mean": 0.9894022941589355, "reward_meter_mean": 0.9894022941589355, "reward_meter_std": 0.007141584530472755, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9894022941589355, "reward_total_composite_std": 0.007141584530472755, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1353.0} {"timestamp_utc": "2026-04-11T22:18:09Z", "mode": "train", "global_step": 1354, "epoch": 0.05228606734630831, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.9e-06, "num_tokens": 2930759.0, "completions/mean_length": 25.0, "completions/min_length": 25.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.0, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.9938231706619263, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9938231706619263, "rewards/total_composite/std": 0.0, "reward": 0.9938231706619263, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0013447451638057828, "sampling/sampling_logp_difference/max": 0.0640781968832016, "sampling/importance_sampling_ratio/min": 0.9379316568374634, "sampling/importance_sampling_ratio/mean": 0.9998536705970764, "sampling/importance_sampling_ratio/max": 1.0308102369308472, "entropy": 0.007196415564976633, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9938231706619263, "reward_meter_mean": 0.9938231706619263, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9938231706619263, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1354.0} {"timestamp_utc": "2026-04-11T22:18:14Z", "mode": "train", "global_step": 1355, "epoch": 0.05232468334877973, "loss": 0.0152, "grad_norm": 4.901665210723877, "learning_rate": 5.8969696969696975e-06, "num_tokens": 2932594.0, "completions/mean_length": 69.375, "completions/min_length": 69.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.375, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9189275503158569, "rewards/meter/std": 0.10219180583953857, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9189275503158569, "rewards/total_composite/std": 0.10219180583953857, "reward": 0.9189275503158569, "reward_std": 0.10219179838895798, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005498027428984642, "sampling/sampling_logp_difference/max": 0.6047682762145996, "sampling/importance_sampling_ratio/min": 0.5462009906768799, "sampling/importance_sampling_ratio/mean": 1.0005221366882324, "sampling/importance_sampling_ratio/max": 1.334370493888855, "entropy": 0.02587844303343445, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/high_mean": 0.0018115942366421223, "clip_ratio/high_max": 0.0018115942366421223, "clip_ratio/region_mean": 0.005283816484734416, "reward_total_mean": 0.9189275503158569, "reward_meter_mean": 0.9189275503158569, "reward_meter_std": 0.10219180583953857, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9189275503158569, "reward_total_composite_std": 0.10219180583953857, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1355.0} {"timestamp_utc": "2026-04-11T22:18:19Z", "mode": "train", "global_step": 1356, "epoch": 0.052363299351251155, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.893939393939394e-06, "num_tokens": 2934330.0, "completions/mean_length": 69.0, "completions/min_length": 69.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9552905559539795, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9552905559539795, "rewards/total_composite/std": 0.0, "reward": 0.9552905559539795, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0015065257903188467, "sampling/sampling_logp_difference/max": 0.050248079001903534, "sampling/importance_sampling_ratio/min": 0.9541056156158447, "sampling/importance_sampling_ratio/mean": 1.0009177923202515, "sampling/importance_sampling_ratio/max": 1.0515319108963013, "entropy": 0.015120051801204681, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9552905559539795, "reward_meter_mean": 0.9552905559539795, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9552905559539795, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1356.0} {"timestamp_utc": "2026-04-11T22:18:25Z", "mode": "train", "global_step": 1357, "epoch": 0.05240191535372258, "loss": -0.0044, "grad_norm": 0.4244149923324585, "learning_rate": 5.890909090909091e-06, "num_tokens": 2936648.0, "completions/mean_length": 105.75, "completions/min_length": 103.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.75, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9937440752983093, "rewards/meter/std": 1.3326747648534365e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9937440752983093, "rewards/total_composite/std": 1.3326747648534365e-05, "reward": 0.9937440752983093, "reward_std": 1.3335940820979886e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013487651012837887, "sampling/sampling_logp_difference/max": 3.749974250793457, "sampling/importance_sampling_ratio/min": 0.023518351837992668, "sampling/importance_sampling_ratio/mean": 0.9970479607582092, "sampling/importance_sampling_ratio/max": 1.9339476823806763, "entropy": 0.01315067190444097, "clip_ratio/low_mean": 0.0036407767329365015, "clip_ratio/low_min": 0.0036407767329365015, "clip_ratio/high_mean": 0.0021929824724793434, "clip_ratio/high_max": 0.0021929824724793434, "clip_ratio/region_mean": 0.005833759205415845, "reward_total_mean": 0.9937440752983093, "reward_meter_mean": 0.9937440752983093, "reward_meter_std": 1.3326747648534365e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9937440752983093, "reward_total_composite_std": 1.3326747648534365e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1357.0} {"timestamp_utc": "2026-04-11T22:18:30Z", "mode": "train", "global_step": 1358, "epoch": 0.052440531356194, "loss": -0.031, "grad_norm": 3.879251003265381, "learning_rate": 5.887878787878788e-06, "num_tokens": 2938422.0, "completions/mean_length": 52.75, "completions/min_length": 52.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.75, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8867310285568237, "rewards/meter/std": 0.022821886464953423, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8867310285568237, "rewards/total_composite/std": 0.022821886464953423, "reward": 0.8867310285568237, "reward_std": 0.02282189205288887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007701834663748741, "sampling/sampling_logp_difference/max": 1.2292070388793945, "sampling/importance_sampling_ratio/min": 0.29252442717552185, "sampling/importance_sampling_ratio/mean": 1.0009299516677856, "sampling/importance_sampling_ratio/max": 1.3026020526885986, "entropy": 0.015542365261353552, "clip_ratio/low_mean": 0.007211538730189204, "clip_ratio/low_min": 0.007211538730189204, "clip_ratio/high_mean": 0.004310344811528921, "clip_ratio/high_max": 0.004310344811528921, "clip_ratio/region_mean": 0.011521883541718125, "reward_total_mean": 0.8867310285568237, "reward_meter_mean": 0.8867310285568237, "reward_meter_std": 0.022821886464953423, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8867310285568237, "reward_total_composite_std": 0.022821886464953423, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1358.0} {"timestamp_utc": "2026-04-11T22:18:35Z", "mode": "train", "global_step": 1359, "epoch": 0.05247914735866543, "loss": -0.0, "grad_norm": 1.0476194620132446, "learning_rate": 5.884848484848486e-06, "num_tokens": 2940174.0, "completions/mean_length": 69.0, "completions/min_length": 69.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9552651643753052, "rewards/meter/std": 4.707126208813861e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9552651643753052, "rewards/total_composite/std": 4.707126208813861e-05, "reward": 0.9552651643753052, "reward_std": 4.707125117420219e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0035026499535888433, "sampling/sampling_logp_difference/max": 0.7407348155975342, "sampling/importance_sampling_ratio/min": 0.4767634868621826, "sampling/importance_sampling_ratio/mean": 0.9997753500938416, "sampling/importance_sampling_ratio/max": 1.0459482669830322, "entropy": 0.017114304471760988, "clip_ratio/low_mean": 0.0018115942366421223, "clip_ratio/low_min": 0.0018115942366421223, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0018115942366421223, "reward_total_mean": 0.9552651643753052, "reward_meter_mean": 0.9552651643753052, "reward_meter_std": 4.707126208813861e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9552651643753052, "reward_total_composite_std": 4.707126208813861e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1359.0} {"timestamp_utc": "2026-04-11T22:18:40Z", "mode": "train", "global_step": 1360, "epoch": 0.05251776336113686, "loss": 0.0645, "grad_norm": 5.233614444732666, "learning_rate": 5.881818181818182e-06, "num_tokens": 2942268.0, "completions/mean_length": 93.75, "completions/min_length": 85.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.75, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9805166125297546, "rewards/meter/std": 0.018798649311065674, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8180726766586304, "rewards/total_composite/std": 0.1804710328578949, "reward": 0.8180726766586304, "reward_std": 0.1804710179567337, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0661487951874733, "sampling/sampling_logp_difference/max": 14.634542465209961, "sampling/importance_sampling_ratio/min": 4.408582583437237e-07, "sampling/importance_sampling_ratio/mean": 1.0017428398132324, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.060703362338244915, "clip_ratio/low_mean": 0.008728962624445558, "clip_ratio/low_min": 0.008728962624445558, "clip_ratio/high_mean": 0.008542029187083244, "clip_ratio/high_max": 0.008542029187083244, "clip_ratio/region_mean": 0.017270991811528802, "reward_total_mean": 0.8180726766586304, "reward_meter_mean": 0.9805166125297546, "reward_meter_std": 0.018798649311065674, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8180726766586304, "reward_total_composite_std": 0.1804710328578949, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1360.0} {"timestamp_utc": "2026-04-11T22:18:46Z", "mode": "train", "global_step": 1361, "epoch": 0.05255637936360828, "loss": 0.0355, "grad_norm": 8.74460506439209, "learning_rate": 5.878787878787879e-06, "num_tokens": 2944566.0, "completions/mean_length": 111.25, "completions/min_length": 106.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.25, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.9977883100509644, "rewards/meter/std": 0.0012346386210992932, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9977883100509644, "rewards/total_composite/std": 0.0012346386210992932, "reward": 0.9977883100509644, "reward_std": 0.0012346338480710983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017775701358914375, "sampling/sampling_logp_difference/max": 3.082744598388672, "sampling/importance_sampling_ratio/min": 0.0458332896232605, "sampling/importance_sampling_ratio/mean": 0.9978323578834534, "sampling/importance_sampling_ratio/max": 1.3729517459869385, "entropy": 0.05511250696144998, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/high_mean": 0.006844450021162629, "clip_ratio/high_max": 0.006844450021162629, "clip_ratio/region_mean": 0.008927783463150263, "reward_total_mean": 0.9977883100509644, "reward_meter_mean": 0.9977883100509644, "reward_meter_std": 0.0012346386210992932, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9977883100509644, "reward_total_composite_std": 0.0012346386210992932, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1361.0} {"timestamp_utc": "2026-04-11T22:18:51Z", "mode": "train", "global_step": 1362, "epoch": 0.05259499536607971, "loss": 0.0049, "grad_norm": 0.3360426425933838, "learning_rate": 5.875757575757576e-06, "num_tokens": 2946263.0, "completions/mean_length": 57.125, "completions/min_length": 57.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.125, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9812615513801575, "rewards/meter/std": 0.03090582601726055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9812615513801575, "rewards/total_composite/std": 0.03090582601726055, "reward": 0.9812615513801575, "reward_std": 0.030905846506357193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0031550773419439793, "sampling/sampling_logp_difference/max": 0.23139214515686035, "sampling/importance_sampling_ratio/min": 0.7934283018112183, "sampling/importance_sampling_ratio/mean": 1.0005500316619873, "sampling/importance_sampling_ratio/max": 1.0982553958892822, "entropy": 0.023121238918974996, "clip_ratio/low_mean": 0.0021551724057644606, "clip_ratio/low_min": 0.0021551724057644606, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0021551724057644606, "reward_total_mean": 0.9812615513801575, "reward_meter_mean": 0.9812615513801575, "reward_meter_std": 0.03090582601726055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9812615513801575, "reward_total_composite_std": 0.03090582601726055, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1362.0} {"timestamp_utc": "2026-04-11T22:18:56Z", "mode": "train", "global_step": 1363, "epoch": 0.05263361136855113, "loss": -0.0001, "grad_norm": 0.08342370390892029, "learning_rate": 5.872727272727273e-06, "num_tokens": 2948514.0, "completions/mean_length": 106.375, "completions/min_length": 106.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.375, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9985930919647217, "rewards/meter/std": 2.823883960445528e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985930919647217, "rewards/total_composite/std": 2.823883960445528e-06, "reward": 0.9985930919647217, "reward_std": 2.8178565116832033e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0024569786619395018, "sampling/sampling_logp_difference/max": 0.7089509963989258, "sampling/importance_sampling_ratio/min": 0.8591918349266052, "sampling/importance_sampling_ratio/mean": 1.00091552734375, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.012435438693501055, "clip_ratio/low_mean": 0.001179245300590992, "clip_ratio/low_min": 0.001179245300590992, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.001179245300590992, "reward_total_mean": 0.9985930919647217, "reward_meter_mean": 0.9985930919647217, "reward_meter_std": 2.823883960445528e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985930919647217, "reward_total_composite_std": 2.823883960445528e-06, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1363.0} {"timestamp_utc": "2026-04-11T22:19:00Z", "mode": "train", "global_step": 1364, "epoch": 0.052672227371022555, "loss": -0.0015, "grad_norm": 0.03341261297464371, "learning_rate": 5.8696969696969694e-06, "num_tokens": 2949933.0, "completions/mean_length": 31.375, "completions/min_length": 31.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.998595118522644, "rewards/meter/std": 8.492683264194056e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.998595118522644, "rewards/total_composite/std": 8.492683264194056e-06, "reward": 0.998595118522644, "reward_std": 8.489590072713327e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005677795968949795, "sampling/sampling_logp_difference/max": 1.178452491760254, "sampling/importance_sampling_ratio/min": 0.30775460600852966, "sampling/importance_sampling_ratio/mean": 0.9981810450553894, "sampling/importance_sampling_ratio/max": 1.0307748317718506, "entropy": 0.0087299186270684, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0036764706019312143, "clip_ratio/high_max": 0.0036764706019312143, "clip_ratio/region_mean": 0.0036764706019312143, "reward_total_mean": 0.998595118522644, "reward_meter_mean": 0.998595118522644, "reward_meter_std": 8.492683264194056e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.998595118522644, "reward_total_composite_std": 8.492683264194056e-06, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1364.0} {"timestamp_utc": "2026-04-11T22:19:11Z", "mode": "train", "global_step": 1365, "epoch": 0.05271084337349398, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.8666666666666675e-06, "num_tokens": 2951381.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.0, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0, "rewards/total_composite/std": 0.0, "reward": 0.0, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.0, "reward_meter_mean": 0.0, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1365.0} {"timestamp_utc": "2026-04-11T22:19:16Z", "mode": "train", "global_step": 1366, "epoch": 0.0527494593759654, "loss": 0.0157, "grad_norm": 1.086145043373108, "learning_rate": 5.863636363636364e-06, "num_tokens": 2953710.0, "completions/mean_length": 124.125, "completions/min_length": 118.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.125, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.008707379922270775, "rewards/meter/std": 0.0006256515043787658, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.005804920103400946, "rewards/total_composite/std": 0.0004171009932179004, "reward": 0.005804920103400946, "reward_std": 0.00041710102232173085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0039140344597399235, "sampling/sampling_logp_difference/max": 1.0388498306274414, "sampling/importance_sampling_ratio/min": 0.3538614511489868, "sampling/importance_sampling_ratio/mean": 0.9994657635688782, "sampling/importance_sampling_ratio/max": 1.1276392936706543, "entropy": 0.013790553552098572, "clip_ratio/low_mean": 0.003000000142492354, "clip_ratio/low_min": 0.003000000142492354, "clip_ratio/high_mean": 0.0010593220358714461, "clip_ratio/high_max": 0.0010593220358714461, "clip_ratio/region_mean": 0.0040593221783638, "reward_total_mean": 0.005804920103400946, "reward_meter_mean": 0.008707379922270775, "reward_meter_std": 0.0006256515043787658, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.005804920103400946, "reward_total_composite_std": 0.0004171009932179004, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1366.0} {"timestamp_utc": "2026-04-11T22:19:22Z", "mode": "train", "global_step": 1367, "epoch": 0.05278807537843683, "loss": 0.0523, "grad_norm": 2.2731573581695557, "learning_rate": 5.860606060606061e-06, "num_tokens": 2956004.0, "completions/mean_length": 121.75, "completions/min_length": 104.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.75, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.009198897518217564, "rewards/meter/std": 0.003665580879896879, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.15430334210395813, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.007003560662269592, "rewards/total_composite/std": 0.004142544232308865, "reward": 0.007003560662269592, "reward_std": 0.004142543766647577, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01424005813896656, "sampling/sampling_logp_difference/max": 2.3622944355010986, "sampling/importance_sampling_ratio/min": 0.09420382976531982, "sampling/importance_sampling_ratio/mean": 0.997249960899353, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.023674575379118323, "clip_ratio/low_mean": 0.0029842520598322153, "clip_ratio/low_min": 0.0029842520598322153, "clip_ratio/high_mean": 0.0024038462433964014, "clip_ratio/high_max": 0.0024038462433964014, "clip_ratio/region_mean": 0.005388098303228617, "reward_total_mean": 0.007003560662269592, "reward_meter_mean": 0.009198897518217564, "reward_meter_std": 0.003665580879896879, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.15430334210395813, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.007003560662269592, "reward_total_composite_std": 0.004142544232308865, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1367.0} {"timestamp_utc": "2026-04-11T22:19:27Z", "mode": "train", "global_step": 1368, "epoch": 0.05282669138090825, "loss": 0.0028, "grad_norm": 4.357656002044678, "learning_rate": 5.8575757575757584e-06, "num_tokens": 2957996.0, "completions/mean_length": 75.0, "completions/min_length": 75.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.0, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.07222011685371399, "rewards/meter/std": 0.0001869234984042123, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.07222011685371399, "rewards/total_composite/std": 0.0001869234984042123, "reward": 0.07222011685371399, "reward_std": 0.0001869234984042123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00786028616130352, "sampling/sampling_logp_difference/max": 0.9992654919624329, "sampling/importance_sampling_ratio/min": 0.3681497573852539, "sampling/importance_sampling_ratio/mean": 1.0020109415054321, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.019569259136915207, "clip_ratio/low_mean": 0.0033333334140479565, "clip_ratio/low_min": 0.0033333334140479565, "clip_ratio/high_mean": 0.006666666828095913, "clip_ratio/high_max": 0.006666666828095913, "clip_ratio/region_mean": 0.01000000024214387, "reward_total_mean": 0.07222011685371399, "reward_meter_mean": 0.07222011685371399, "reward_meter_std": 0.0001869234984042123, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.07222011685371399, "reward_total_composite_std": 0.0001869234984042123, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1368.0} {"timestamp_utc": "2026-04-11T22:19:32Z", "mode": "train", "global_step": 1369, "epoch": 0.052865307383379675, "loss": 0.0029, "grad_norm": 3.9614062309265137, "learning_rate": 5.854545454545455e-06, "num_tokens": 2959718.0, "completions/mean_length": 57.25, "completions/min_length": 57.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.25, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9843666553497314, "rewards/meter/std": 0.019942112267017365, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9843666553497314, "rewards/total_composite/std": 0.019942112267017365, "reward": 0.9843666553497314, "reward_std": 0.019942117854952812, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014284457080066204, "sampling/sampling_logp_difference/max": 1.651278018951416, "sampling/importance_sampling_ratio/min": 0.19180461764335632, "sampling/importance_sampling_ratio/mean": 1.0002416372299194, "sampling/importance_sampling_ratio/max": 1.5830457210540771, "entropy": 0.04951313021592796, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.008771929889917374, "clip_ratio/high_max": 0.008771929889917374, "clip_ratio/region_mean": 0.013082274701446295, "reward_total_mean": 0.9843666553497314, "reward_meter_mean": 0.9843666553497314, "reward_meter_std": 0.019942112267017365, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9843666553497314, "reward_total_composite_std": 0.019942112267017365, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1369.0} {"timestamp_utc": "2026-04-11T22:19:37Z", "mode": "train", "global_step": 1370, "epoch": 0.0529039233858511, "loss": -0.0057, "grad_norm": 8.277812004089355, "learning_rate": 5.851515151515152e-06, "num_tokens": 2961390.0, "completions/mean_length": 60.0, "completions/min_length": 58.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8910601139068604, "rewards/meter/std": 0.3009309470653534, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8910601139068604, "rewards/total_composite/std": 0.3009309470653534, "reward": 0.8910601139068604, "reward_std": 0.3009309470653534, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01927165687084198, "sampling/sampling_logp_difference/max": 1.9251534938812256, "sampling/importance_sampling_ratio/min": 0.14585337042808533, "sampling/importance_sampling_ratio/mean": 0.9998040795326233, "sampling/importance_sampling_ratio/max": 1.9132729768753052, "entropy": 0.034557814709842205, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.012435226701200008, "clip_ratio/high_max": 0.012435226701200008, "clip_ratio/region_mean": 0.012435226701200008, "reward_total_mean": 0.8910601139068604, "reward_meter_mean": 0.8910601139068604, "reward_meter_std": 0.3009309470653534, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8910601139068604, "reward_total_composite_std": 0.3009309470653534, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1370.0} {"timestamp_utc": "2026-04-11T22:19:42Z", "mode": "train", "global_step": 1371, "epoch": 0.052942539388322524, "loss": 0.0062, "grad_norm": 1.4612329006195068, "learning_rate": 5.8484848484848485e-06, "num_tokens": 2963401.0, "completions/mean_length": 90.375, "completions/min_length": 90.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.375, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9940619468688965, "rewards/meter/std": 0.00030107275233604014, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9940619468688965, "rewards/total_composite/std": 0.00030107275233604014, "reward": 0.9940619468688965, "reward_std": 0.0003010733926203102, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01160053163766861, "sampling/sampling_logp_difference/max": 2.3534936904907227, "sampling/importance_sampling_ratio/min": 0.09503655135631561, "sampling/importance_sampling_ratio/mean": 0.9960593581199646, "sampling/importance_sampling_ratio/max": 1.3237727880477905, "entropy": 0.017025562352500856, "clip_ratio/low_mean": 0.00412087922450155, "clip_ratio/low_min": 0.00412087922450155, "clip_ratio/high_mean": 0.0055555556900799274, "clip_ratio/high_max": 0.0055555556900799274, "clip_ratio/region_mean": 0.009676434914581478, "reward_total_mean": 0.9940619468688965, "reward_meter_mean": 0.9940619468688965, "reward_meter_std": 0.00030107275233604014, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9940619468688965, "reward_total_composite_std": 0.00030107275233604014, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1371.0} {"timestamp_utc": "2026-04-11T22:19:47Z", "mode": "train", "global_step": 1372, "epoch": 0.05298115539079395, "loss": -0.0022, "grad_norm": 1.12401282787323, "learning_rate": 5.845454545454547e-06, "num_tokens": 2965147.0, "completions/mean_length": 63.25, "completions/min_length": 61.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.998638927936554, "rewards/meter/std": 7.231163181131706e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.998638927936554, "rewards/total_composite/std": 7.231163181131706e-05, "reward": 0.998638927936554, "reward_std": 7.229840412037447e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005806723143905401, "sampling/sampling_logp_difference/max": 0.5715174674987793, "sampling/importance_sampling_ratio/min": 0.5646678805351257, "sampling/importance_sampling_ratio/mean": 0.9998002052307129, "sampling/importance_sampling_ratio/max": 1.3009048700332642, "entropy": 0.02991713653318584, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/region_mean": 0.005890377098694444, "reward_total_mean": 0.998638927936554, "reward_meter_mean": 0.998638927936554, "reward_meter_std": 7.231163181131706e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.998638927936554, "reward_total_composite_std": 7.231163181131706e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1372.0} {"timestamp_utc": "2026-04-11T22:19:51Z", "mode": "train", "global_step": 1373, "epoch": 0.05301977139326537, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.842424242424243e-06, "num_tokens": 2967027.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9974405169487, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9974405169487, "rewards/total_composite/std": 0.0, "reward": 0.9974405169487, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0007480629137717187, "sampling/sampling_logp_difference/max": 0.020914193242788315, "sampling/importance_sampling_ratio/min": 0.9986199736595154, "sampling/importance_sampling_ratio/mean": 1.0007416009902954, "sampling/importance_sampling_ratio/max": 1.021134376525879, "entropy": 0.005566546169575304, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9974405169487, "reward_meter_mean": 0.9974405169487, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9974405169487, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1373.0} {"timestamp_utc": "2026-04-11T22:20:01Z", "mode": "train", "global_step": 1374, "epoch": 0.053058387395736796, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.83939393939394e-06, "num_tokens": 2968219.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.0, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0, "rewards/total_composite/std": 0.0, "reward": 0.0, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.0, "reward_meter_mean": 0.0, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1374.0} {"timestamp_utc": "2026-04-11T22:20:07Z", "mode": "train", "global_step": 1375, "epoch": 0.05309700339820822, "loss": -0.0176, "grad_norm": 0.45738545060157776, "learning_rate": 5.836363636363637e-06, "num_tokens": 2970321.0, "completions/mean_length": 103.75, "completions/min_length": 103.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.75, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.957053542137146, "rewards/meter/std": 0.0052316258661448956, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.957053542137146, "rewards/total_composite/std": 0.0052316258661448956, "reward": 0.957053542137146, "reward_std": 0.005231638438999653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0026220884174108505, "sampling/sampling_logp_difference/max": 0.3043633699417114, "sampling/importance_sampling_ratio/min": 0.7375928163528442, "sampling/importance_sampling_ratio/mean": 0.9997009038925171, "sampling/importance_sampling_ratio/max": 1.1372449398040771, "entropy": 0.013364620739594102, "clip_ratio/low_mean": 0.0012135922443121672, "clip_ratio/low_min": 0.0012135922443121672, "clip_ratio/high_mean": 0.0011467889416962862, "clip_ratio/high_max": 0.0011467889416962862, "clip_ratio/region_mean": 0.0023603811860084534, "reward_total_mean": 0.957053542137146, "reward_meter_mean": 0.957053542137146, "reward_meter_std": 0.0052316258661448956, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.957053542137146, "reward_total_composite_std": 0.0052316258661448956, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1375.0} {"timestamp_utc": "2026-04-11T22:20:12Z", "mode": "train", "global_step": 1376, "epoch": 0.053135619400679644, "loss": 0.0056, "grad_norm": 0.9059351682662964, "learning_rate": 5.833333333333334e-06, "num_tokens": 2972032.0, "completions/mean_length": 68.875, "completions/min_length": 68.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9553878307342529, "rewards/meter/std": 0.00027519784634932876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9553878307342529, "rewards/total_composite/std": 0.00027519784634932876, "reward": 0.9553878307342529, "reward_std": 0.0002752068976406008, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002282181289047003, "sampling/sampling_logp_difference/max": 0.5202398300170898, "sampling/importance_sampling_ratio/min": 0.5943779945373535, "sampling/importance_sampling_ratio/mean": 1.0003669261932373, "sampling/importance_sampling_ratio/max": 1.0516102313995361, "entropy": 0.01157037157099694, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9553878307342529, "reward_meter_mean": 0.9553878307342529, "reward_meter_std": 0.00027519784634932876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9553878307342529, "reward_total_composite_std": 0.00027519784634932876, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1376.0} {"timestamp_utc": "2026-04-11T22:20:17Z", "mode": "train", "global_step": 1377, "epoch": 0.05317423540315107, "loss": -0.0299, "grad_norm": 2.9744873046875, "learning_rate": 5.83030303030303e-06, "num_tokens": 2974346.0, "completions/mean_length": 105.25, "completions/min_length": 90.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.25, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.19511528313159943, "rewards/meter/std": 0.3402114808559418, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.19511528313159943, "rewards/total_composite/std": 0.3402114808559418, "reward": 0.19511528313159943, "reward_std": 0.3402114510536194, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017067207023501396, "sampling/sampling_logp_difference/max": 2.2724545001983643, "sampling/importance_sampling_ratio/min": 0.10305891185998917, "sampling/importance_sampling_ratio/mean": 1.0021616220474243, "sampling/importance_sampling_ratio/max": 1.5982531309127808, "entropy": 0.07001882325857878, "clip_ratio/low_mean": 0.01067229371983558, "clip_ratio/low_min": 0.01067229371983558, "clip_ratio/high_mean": 0.006797706708312035, "clip_ratio/high_max": 0.006797706708312035, "clip_ratio/region_mean": 0.017470000428147614, "reward_total_mean": 0.19511528313159943, "reward_meter_mean": 0.19511528313159943, "reward_meter_std": 0.3402114808559418, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.19511528313159943, "reward_total_composite_std": 0.3402114808559418, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1377.0} {"timestamp_utc": "2026-04-11T22:20:21Z", "mode": "train", "global_step": 1378, "epoch": 0.05321285140562249, "loss": -0.0152, "grad_norm": 14.091780662536621, "learning_rate": 5.8272727272727285e-06, "num_tokens": 2975937.0, "completions/mean_length": 36.875, "completions/min_length": 36.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.6659175157546997, "rewards/meter/std": 0.18820102512836456, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6659175157546997, "rewards/total_composite/std": 0.18820102512836456, "reward": 0.6659175157546997, "reward_std": 0.18820102512836456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03318372741341591, "sampling/sampling_logp_difference/max": 1.3043758869171143, "sampling/importance_sampling_ratio/min": 0.27134186029434204, "sampling/importance_sampling_ratio/mean": 0.9888492822647095, "sampling/importance_sampling_ratio/max": 1.4609684944152832, "entropy": 0.10958861047402024, "clip_ratio/low_mean": 0.010322822956368327, "clip_ratio/low_min": 0.010322822956368327, "clip_ratio/high_mean": 0.01609078561887145, "clip_ratio/high_max": 0.01609078561887145, "clip_ratio/region_mean": 0.026413608575239778, "reward_total_mean": 0.6659175157546997, "reward_meter_mean": 0.6659175157546997, "reward_meter_std": 0.18820102512836456, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6659175157546997, "reward_total_composite_std": 0.18820102512836456, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1378.0} {"timestamp_utc": "2026-04-11T22:20:26Z", "mode": "train", "global_step": 1379, "epoch": 0.05325146740809392, "loss": 0.0108, "grad_norm": 4.158636093139648, "learning_rate": 5.824242424242425e-06, "num_tokens": 2977824.0, "completions/mean_length": 67.875, "completions/min_length": 66.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9594380855560303, "rewards/meter/std": 0.009847787208855152, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9594380855560303, "rewards/total_composite/std": 0.009847787208855152, "reward": 0.9594380855560303, "reward_std": 0.009847790002822876, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006572279147803783, "sampling/sampling_logp_difference/max": 0.49502575397491455, "sampling/importance_sampling_ratio/min": 0.609555184841156, "sampling/importance_sampling_ratio/mean": 1.003727912902832, "sampling/importance_sampling_ratio/max": 1.5699145793914795, "entropy": 0.04251991002820432, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/region_mean": 0.0018939394503831863, "reward_total_mean": 0.9594380855560303, "reward_meter_mean": 0.9594380855560303, "reward_meter_std": 0.009847787208855152, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9594380855560303, "reward_total_composite_std": 0.009847787208855152, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1379.0} {"timestamp_utc": "2026-04-11T22:20:31Z", "mode": "train", "global_step": 1380, "epoch": 0.05329008341056534, "loss": -0.0183, "grad_norm": 6.115478992462158, "learning_rate": 5.821212121212122e-06, "num_tokens": 2979769.0, "completions/mean_length": 68.125, "completions/min_length": 66.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.0579112246632576, "rewards/meter/std": 0.04924581199884415, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0579112246632576, "rewards/total_composite/std": 0.04924581199884415, "reward": 0.0579112246632576, "reward_std": 0.04924580827355385, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020362965762615204, "sampling/sampling_logp_difference/max": 1.2489221096038818, "sampling/importance_sampling_ratio/min": 0.28681376576423645, "sampling/importance_sampling_ratio/mean": 1.0007599592208862, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08677392546087503, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.007194617064669728, "clip_ratio/high_max": 0.007194617064669728, "clip_ratio/region_mean": 0.009088556515052915, "reward_total_mean": 0.0579112246632576, "reward_meter_mean": 0.0579112246632576, "reward_meter_std": 0.04924581199884415, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0579112246632576, "reward_total_composite_std": 0.04924581199884415, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1380.0} {"timestamp_utc": "2026-04-11T22:20:37Z", "mode": "train", "global_step": 1381, "epoch": 0.053328699413036765, "loss": 0.015, "grad_norm": 4.74990177154541, "learning_rate": 5.8181818181818185e-06, "num_tokens": 2981914.0, "completions/mean_length": 108.125, "completions/min_length": 100.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.125, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.5881621837615967, "rewards/meter/std": 0.35498878359794617, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5881621837615967, "rewards/total_composite/std": 0.35498878359794617, "reward": 0.5881621837615967, "reward_std": 0.35498878359794617, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03959234058856964, "sampling/sampling_logp_difference/max": 3.609257698059082, "sampling/importance_sampling_ratio/min": 0.027071936056017876, "sampling/importance_sampling_ratio/mean": 1.0022164583206177, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11689508147537708, "clip_ratio/low_mean": 0.01517009362578392, "clip_ratio/low_min": 0.01517009362578392, "clip_ratio/high_mean": 0.016357550164684653, "clip_ratio/high_max": 0.016357550164684653, "clip_ratio/region_mean": 0.031527643790468574, "reward_total_mean": 0.5881621837615967, "reward_meter_mean": 0.5881621837615967, "reward_meter_std": 0.35498878359794617, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5881621837615967, "reward_total_composite_std": 0.35498878359794617, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1381.0} {"timestamp_utc": "2026-04-11T22:20:42Z", "mode": "train", "global_step": 1382, "epoch": 0.05336731541550819, "loss": 0.0115, "grad_norm": 5.238241672515869, "learning_rate": 5.815151515151516e-06, "num_tokens": 2984309.0, "completions/mean_length": 110.375, "completions/min_length": 109.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.375, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9307272434234619, "rewards/meter/std": 0.06542084366083145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9307272434234619, "rewards/total_composite/std": 0.06542084366083145, "reward": 0.9307272434234619, "reward_std": 0.06542082875967026, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011836927384138107, "sampling/sampling_logp_difference/max": 1.586113452911377, "sampling/importance_sampling_ratio/min": 0.20471970736980438, "sampling/importance_sampling_ratio/mean": 1.0022673606872559, "sampling/importance_sampling_ratio/max": 1.8693678379058838, "entropy": 0.03451790288090706, "clip_ratio/low_mean": 0.0022522523067891598, "clip_ratio/low_min": 0.0022522523067891598, "clip_ratio/high_mean": 0.0045456422958523035, "clip_ratio/high_max": 0.0045456422958523035, "clip_ratio/region_mean": 0.006797894602641463, "reward_total_mean": 0.9307272434234619, "reward_meter_mean": 0.9307272434234619, "reward_meter_std": 0.06542084366083145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9307272434234619, "reward_total_composite_std": 0.06542084366083145, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1382.0} {"timestamp_utc": "2026-04-11T22:20:47Z", "mode": "train", "global_step": 1383, "epoch": 0.05340593141797961, "loss": -0.0076, "grad_norm": 2.648869514465332, "learning_rate": 5.812121212121212e-06, "num_tokens": 2986145.0, "completions/mean_length": 72.5, "completions/min_length": 72.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.5, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.16342562437057495, "rewards/meter/std": 0.02146931178867817, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.16342562437057495, "rewards/total_composite/std": 0.02146931178867817, "reward": 0.16342562437057495, "reward_std": 0.021469315513968468, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010640337131917477, "sampling/sampling_logp_difference/max": 1.0853865146636963, "sampling/importance_sampling_ratio/min": 0.3377712070941925, "sampling/importance_sampling_ratio/mean": 0.9983828663825989, "sampling/importance_sampling_ratio/max": 1.2706122398376465, "entropy": 0.047764832619577646, "clip_ratio/low_mean": 0.003448439878411591, "clip_ratio/low_min": 0.003448439878411591, "clip_ratio/high_mean": 0.0033333334140479565, "clip_ratio/high_max": 0.0033333334140479565, "clip_ratio/region_mean": 0.0067817732924595475, "reward_total_mean": 0.16342562437057495, "reward_meter_mean": 0.16342562437057495, "reward_meter_std": 0.02146931178867817, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.16342562437057495, "reward_total_composite_std": 0.02146931178867817, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1383.0} {"timestamp_utc": "2026-04-11T22:20:53Z", "mode": "train", "global_step": 1384, "epoch": 0.05344454742045104, "loss": 0.0208, "grad_norm": 1.0995887517929077, "learning_rate": 5.8090909090909095e-06, "num_tokens": 2988676.0, "completions/mean_length": 139.375, "completions/min_length": 124.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.375, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9977383017539978, "rewards/meter/std": 0.0004581170796882361, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9977383017539978, "rewards/total_composite/std": 0.0004581170796882361, "reward": 0.9977383017539978, "reward_std": 0.0004581264511216432, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011376053094863892, "sampling/sampling_logp_difference/max": 1.0027985572814941, "sampling/importance_sampling_ratio/min": 0.3668513596057892, "sampling/importance_sampling_ratio/mean": 1.0013431310653687, "sampling/importance_sampling_ratio/max": 1.842280387878418, "entropy": 0.049101441632956266, "clip_ratio/low_mean": 0.00439525255933404, "clip_ratio/low_min": 0.00439525255933404, "clip_ratio/high_mean": 0.007183039328083396, "clip_ratio/high_max": 0.007183039328083396, "clip_ratio/region_mean": 0.011578291887417436, "reward_total_mean": 0.9977383017539978, "reward_meter_mean": 0.9977383017539978, "reward_meter_std": 0.0004581170796882361, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9977383017539978, "reward_total_composite_std": 0.0004581170796882361, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1384.0} {"timestamp_utc": "2026-04-11T22:20:58Z", "mode": "train", "global_step": 1385, "epoch": 0.05348316342292246, "loss": 0.0066, "grad_norm": 1.3943383693695068, "learning_rate": 5.806060606060606e-06, "num_tokens": 2990741.0, "completions/mean_length": 97.125, "completions/min_length": 94.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.125, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9984763860702515, "rewards/meter/std": 0.00016384755144827068, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984763860702515, "rewards/total_composite/std": 0.00016384755144827068, "reward": 0.9984763860702515, "reward_std": 0.00016384724585805088, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007108251564204693, "sampling/sampling_logp_difference/max": 0.9237980842590332, "sampling/importance_sampling_ratio/min": 0.6692610383033752, "sampling/importance_sampling_ratio/mean": 1.0039395093917847, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03119822940789163, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.002618446946144104, "clip_ratio/high_max": 0.002618446946144104, "clip_ratio/region_mean": 0.002618446946144104, "reward_total_mean": 0.9984763860702515, "reward_meter_mean": 0.9984763860702515, "reward_meter_std": 0.00016384755144827068, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984763860702515, "reward_total_composite_std": 0.00016384755144827068, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1385.0} {"timestamp_utc": "2026-04-11T22:21:02Z", "mode": "train", "global_step": 1386, "epoch": 0.053521779425393885, "loss": 0.0083, "grad_norm": 2.2268850803375244, "learning_rate": 5.803030303030304e-06, "num_tokens": 2992484.0, "completions/mean_length": 51.875, "completions/min_length": 51.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.993849515914917, "rewards/meter/std": 0.000359718018444255, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.993849515914917, "rewards/total_composite/std": 0.000359718018444255, "reward": 0.993849515914917, "reward_std": 0.0003597206377889961, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006365790963172913, "sampling/sampling_logp_difference/max": 1.4584407806396484, "sampling/importance_sampling_ratio/min": 0.2325986623764038, "sampling/importance_sampling_ratio/mean": 0.9977045655250549, "sampling/importance_sampling_ratio/max": 1.168224573135376, "entropy": 0.01870046101976186, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0024509804788976908, "clip_ratio/high_max": 0.0024509804788976908, "clip_ratio/region_mean": 0.0024509804788976908, "reward_total_mean": 0.993849515914917, "reward_meter_mean": 0.993849515914917, "reward_meter_std": 0.000359718018444255, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.993849515914917, "reward_total_composite_std": 0.000359718018444255, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1386.0} {"timestamp_utc": "2026-04-11T22:21:07Z", "mode": "train", "global_step": 1387, "epoch": 0.05356039542786531, "loss": -0.0231, "grad_norm": 5.518721580505371, "learning_rate": 5.8e-06, "num_tokens": 2994316.0, "completions/mean_length": 71.0, "completions/min_length": 67.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.47317761182785034, "rewards/meter/std": 0.20058581233024597, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.47317761182785034, "rewards/total_composite/std": 0.20058581233024597, "reward": 0.47317761182785034, "reward_std": 0.20058582723140717, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021350044757127762, "sampling/sampling_logp_difference/max": 1.145082950592041, "sampling/importance_sampling_ratio/min": 0.3181975185871124, "sampling/importance_sampling_ratio/mean": 1.0034959316253662, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.046088060829788446, "clip_ratio/low_mean": 0.01254450436681509, "clip_ratio/low_min": 0.01254450436681509, "clip_ratio/high_mean": 0.005000000121071935, "clip_ratio/high_max": 0.005000000121071935, "clip_ratio/region_mean": 0.017544504487887025, "reward_total_mean": 0.47317761182785034, "reward_meter_mean": 0.47317761182785034, "reward_meter_std": 0.20058581233024597, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.47317761182785034, "reward_total_composite_std": 0.20058581233024597, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1387.0} {"timestamp_utc": "2026-04-11T22:21:12Z", "mode": "train", "global_step": 1388, "epoch": 0.053599011430336733, "loss": 0.0005, "grad_norm": 0.10075599700212479, "learning_rate": 5.796969696969698e-06, "num_tokens": 2996219.0, "completions/mean_length": 60.875, "completions/min_length": 60.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9974446892738342, "rewards/meter/std": 1.180111758003477e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9974446892738342, "rewards/total_composite/std": 1.180111758003477e-05, "reward": 0.9974446892738342, "reward_std": 1.180111758003477e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0012331139296293259, "sampling/sampling_logp_difference/max": 0.10192450881004333, "sampling/importance_sampling_ratio/min": 0.9030977487564087, "sampling/importance_sampling_ratio/mean": 1.0004630088806152, "sampling/importance_sampling_ratio/max": 1.0325435400009155, "entropy": 0.010055922786705196, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0020833334419876337, "clip_ratio/high_max": 0.0020833334419876337, "clip_ratio/region_mean": 0.0020833334419876337, "reward_total_mean": 0.9974446892738342, "reward_meter_mean": 0.9974446892738342, "reward_meter_std": 1.180111758003477e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9974446892738342, "reward_total_composite_std": 1.180111758003477e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1388.0} {"timestamp_utc": "2026-04-11T22:21:16Z", "mode": "train", "global_step": 1389, "epoch": 0.05363762743280816, "loss": 0.0835, "grad_norm": 8.576009750366211, "learning_rate": 5.793939393939394e-06, "num_tokens": 2998107.0, "completions/mean_length": 59.0, "completions/min_length": 57.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.0035445806570351124, "rewards/meter/std": 0.0014301573392003775, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.0035445806570351124, "rewards/total_composite/std": 0.0014301573392003775, "reward": 0.0035445806570351124, "reward_std": 0.0014301571063697338, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024933379143476486, "sampling/sampling_logp_difference/max": 1.5268127918243408, "sampling/importance_sampling_ratio/min": 0.21722690761089325, "sampling/importance_sampling_ratio/mean": 0.9965850114822388, "sampling/importance_sampling_ratio/max": 1.513292670249939, "entropy": 0.06611403496935964, "clip_ratio/low_mean": 0.0051369862630963326, "clip_ratio/low_min": 0.0051369862630963326, "clip_ratio/high_mean": 0.01315789483487606, "clip_ratio/high_max": 0.01315789483487606, "clip_ratio/region_mean": 0.018294881097972393, "reward_total_mean": 0.0035445806570351124, "reward_meter_mean": 0.0035445806570351124, "reward_meter_std": 0.0014301573392003775, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.0035445806570351124, "reward_total_composite_std": 0.0014301573392003775, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1389.0} {"timestamp_utc": "2026-04-11T22:21:21Z", "mode": "train", "global_step": 1390, "epoch": 0.05367624343527958, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.790909090909091e-06, "num_tokens": 2999843.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9974405169487, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9974405169487, "rewards/total_composite/std": 0.0, "reward": 0.9974405169487, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0008587094489485025, "sampling/sampling_logp_difference/max": 0.04290404170751572, "sampling/importance_sampling_ratio/min": 0.9952344298362732, "sampling/importance_sampling_ratio/mean": 1.0008423328399658, "sampling/importance_sampling_ratio/max": 1.0438377857208252, "entropy": 0.008749772619921714, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9974405169487, "reward_meter_mean": 0.9974405169487, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9974405169487, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1390.0} {"timestamp_utc": "2026-04-11T22:21:25Z", "mode": "train", "global_step": 1391, "epoch": 0.053714859437751006, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.787878787878788e-06, "num_tokens": 3001475.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9974405169487, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9974405169487, "rewards/total_composite/std": 0.0, "reward": 0.9974405169487, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006601462955586612, "sampling/sampling_logp_difference/max": 0.034192606806755066, "sampling/importance_sampling_ratio/min": 0.9663853645324707, "sampling/importance_sampling_ratio/mean": 1.0004802942276, "sampling/importance_sampling_ratio/max": 1.014174222946167, "entropy": 0.006050599738955498, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9974405169487, "reward_meter_mean": 0.9974405169487, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9974405169487, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1391.0} {"timestamp_utc": "2026-04-11T22:21:31Z", "mode": "train", "global_step": 1392, "epoch": 0.05375347544022243, "loss": -0.0206, "grad_norm": 2.9132845401763916, "learning_rate": 5.784848484848486e-06, "num_tokens": 3003587.0, "completions/mean_length": 102.0, "completions/min_length": 86.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.0, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.9944596290588379, "rewards/meter/std": 0.0061253029853105545, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9944596290588379, "rewards/total_composite/std": 0.0061253029853105545, "reward": 0.9944596290588379, "reward_std": 0.006125311367213726, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026021337136626244, "sampling/sampling_logp_difference/max": 1.339212417602539, "sampling/importance_sampling_ratio/min": 0.2620519697666168, "sampling/importance_sampling_ratio/mean": 0.9999955296516418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11673475662246346, "clip_ratio/low_mean": 0.01127461635041982, "clip_ratio/low_min": 0.01127461635041982, "clip_ratio/high_mean": 0.020271025830879807, "clip_ratio/high_max": 0.020271025830879807, "clip_ratio/region_mean": 0.03154564218129963, "reward_total_mean": 0.9944596290588379, "reward_meter_mean": 0.9944596290588379, "reward_meter_std": 0.0061253029853105545, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9944596290588379, "reward_total_composite_std": 0.0061253029853105545, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1392.0} {"timestamp_utc": "2026-04-11T22:21:38Z", "mode": "train", "global_step": 1393, "epoch": 0.053792091442693854, "loss": -0.0198, "grad_norm": 2.6163511276245117, "learning_rate": 5.781818181818181e-06, "num_tokens": 3007222.0, "completions/mean_length": 241.375, "completions/min_length": 216.0, "completions/max_length": 247.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 241.375, "completions/min_terminated_length": 216.0, "completions/max_terminated_length": 247.0, "rewards/meter/mean": 0.03249666839838028, "rewards/meter/std": 0.018390465527772903, "rewards/count_adherence/mean": 0.7777777910232544, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.025275185704231262, "rewards/total_composite/std": 0.014303695410490036, "reward": 0.025275185704231262, "reward_std": 0.014303693547844887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0122007355093956, "sampling/sampling_logp_difference/max": 5.017805099487305, "sampling/importance_sampling_ratio/min": 0.006619038991630077, "sampling/importance_sampling_ratio/mean": 1.000434160232544, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.024340001633390784, "clip_ratio/low_mean": 0.0016032938146963716, "clip_ratio/low_min": 0.0016032938146963716, "clip_ratio/high_mean": 0.0066431419108994305, "clip_ratio/high_max": 0.0066431419108994305, "clip_ratio/region_mean": 0.008246435725595802, "reward_total_mean": 0.025275185704231262, "reward_meter_mean": 0.03249666839838028, "reward_meter_std": 0.018390465527772903, "reward_count_adherence_mean": 0.7777777910232544, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.025275185704231262, "reward_total_composite_std": 0.014303695410490036, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1393.0} {"timestamp_utc": "2026-04-11T22:21:46Z", "mode": "train", "global_step": 1394, "epoch": 0.05383070744516528, "loss": -0.0229, "grad_norm": 1.1479768753051758, "learning_rate": 5.7787878787878795e-06, "num_tokens": 3011605.0, "completions/mean_length": 349.875, "completions/min_length": 311.0, "completions/max_length": 366.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 349.875, "completions/min_terminated_length": 311.0, "completions/max_terminated_length": 366.0, "rewards/meter/mean": 0.736499547958374, "rewards/meter/std": 0.314837783575058, "rewards/count_adherence/mean": 0.862500011920929, "rewards/count_adherence/std": 0.0517548993229866, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6362034678459167, "rewards/total_composite/std": 0.28191813826560974, "reward": 0.6362034678459167, "reward_std": 0.28191813826560974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014292136766016483, "sampling/sampling_logp_difference/max": 4.043093681335449, "sampling/importance_sampling_ratio/min": 0.017543114721775055, "sampling/importance_sampling_ratio/mean": 1.0002002716064453, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03978452831506729, "clip_ratio/low_mean": 0.004758977622259408, "clip_ratio/low_min": 0.004758977622259408, "clip_ratio/high_mean": 0.009103724150918424, "clip_ratio/high_max": 0.009103724150918424, "clip_ratio/region_mean": 0.013862701773177832, "reward_total_mean": 0.6362034678459167, "reward_meter_mean": 0.736499547958374, "reward_meter_std": 0.314837783575058, "reward_count_adherence_mean": 0.862500011920929, "reward_count_adherence_std": 0.0517548993229866, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6362034678459167, "reward_total_composite_std": 0.28191813826560974, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1394.0} {"timestamp_utc": "2026-04-11T22:21:54Z", "mode": "train", "global_step": 1395, "epoch": 0.0538693234476367, "loss": -0.0125, "grad_norm": 1.5230201482772827, "learning_rate": 5.775757575757577e-06, "num_tokens": 3016186.0, "completions/mean_length": 353.625, "completions/min_length": 331.0, "completions/max_length": 369.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 353.625, "completions/min_terminated_length": 331.0, "completions/max_terminated_length": 369.0, "rewards/meter/mean": 0.33540937304496765, "rewards/meter/std": 0.37437373399734497, "rewards/count_adherence/mean": 0.8181818127632141, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.2744258642196655, "rewards/total_composite/std": 0.30630579590797424, "reward": 0.2744258642196655, "reward_std": 0.30630579590797424, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01354670338332653, "sampling/sampling_logp_difference/max": 3.202211856842041, "sampling/importance_sampling_ratio/min": 0.040672142058610916, "sampling/importance_sampling_ratio/mean": 0.9981418251991272, "sampling/importance_sampling_ratio/max": 1.9025962352752686, "entropy": 0.037523993756622076, "clip_ratio/low_mean": 0.004980219528079033, "clip_ratio/low_min": 0.004980219528079033, "clip_ratio/high_mean": 0.004173983179498464, "clip_ratio/high_max": 0.004173983179498464, "clip_ratio/region_mean": 0.009154202707577497, "reward_total_mean": 0.2744258642196655, "reward_meter_mean": 0.33540937304496765, "reward_meter_std": 0.37437373399734497, "reward_count_adherence_mean": 0.8181818127632141, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.2744258642196655, "reward_total_composite_std": 0.30630579590797424, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1395.0} {"timestamp_utc": "2026-04-11T22:21:59Z", "mode": "train", "global_step": 1396, "epoch": 0.053907939450108126, "loss": -0.0169, "grad_norm": 9.21571159362793, "learning_rate": 5.772727272727273e-06, "num_tokens": 3018115.0, "completions/mean_length": 80.125, "completions/min_length": 74.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.125, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.7805598974227905, "rewards/meter/std": 0.39566588401794434, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7805598974227905, "rewards/total_composite/std": 0.39566588401794434, "reward": 0.7805598974227905, "reward_std": 0.39566585421562195, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035405561327934265, "sampling/sampling_logp_difference/max": 3.599158525466919, "sampling/importance_sampling_ratio/min": 0.027346724644303322, "sampling/importance_sampling_ratio/mean": 0.9964043498039246, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0896854973398149, "clip_ratio/low_mean": 0.0016025641234591603, "clip_ratio/low_min": 0.0016025641234591603, "clip_ratio/high_mean": 0.015528549440205097, "clip_ratio/high_max": 0.015528549440205097, "clip_ratio/region_mean": 0.017131113563664258, "reward_total_mean": 0.7805598974227905, "reward_meter_mean": 0.7805598974227905, "reward_meter_std": 0.39566588401794434, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7805598974227905, "reward_total_composite_std": 0.39566588401794434, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1396.0} {"timestamp_utc": "2026-04-11T22:22:10Z", "mode": "train", "global_step": 1397, "epoch": 0.05394655545257955, "loss": 0.0034, "grad_norm": 1.004015326499939, "learning_rate": 5.76969696969697e-06, "num_tokens": 3024042.0, "completions/mean_length": 488.875, "completions/min_length": 471.0, "completions/max_length": 500.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 488.875, "completions/min_terminated_length": 471.0, "completions/max_terminated_length": 500.0, "rewards/meter/mean": 0.6645517945289612, "rewards/meter/std": 0.2341638058423996, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.047245558351278305, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5220382213592529, "rewards/total_composite/std": 0.1868124008178711, "reward": 0.5220382213592529, "reward_std": 0.1868124008178711, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00705917552113533, "sampling/sampling_logp_difference/max": 1.7648948431015015, "sampling/importance_sampling_ratio/min": 0.17120479047298431, "sampling/importance_sampling_ratio/mean": 1.000388264656067, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.027735163923352957, "clip_ratio/low_mean": 0.0030229381227400154, "clip_ratio/low_min": 0.0030229381227400154, "clip_ratio/high_mean": 0.004108848108444363, "clip_ratio/high_max": 0.004108848108444363, "clip_ratio/region_mean": 0.007131786231184378, "reward_total_mean": 0.5220382213592529, "reward_meter_mean": 0.6645517945289612, "reward_meter_std": 0.2341638058423996, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.047245558351278305, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5220382213592529, "reward_total_composite_std": 0.1868124008178711, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1397.0} {"timestamp_utc": "2026-04-11T22:22:14Z", "mode": "train", "global_step": 1398, "epoch": 0.053985171455050975, "loss": -0.0042, "grad_norm": 9.139034271240234, "learning_rate": 5.766666666666667e-06, "num_tokens": 3025528.0, "completions/mean_length": 43.75, "completions/min_length": 36.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.840766429901123, "rewards/meter/std": 0.2658155858516693, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.840766429901123, "rewards/total_composite/std": 0.2658155858516693, "reward": 0.840766429901123, "reward_std": 0.2658155858516693, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07685981690883636, "sampling/sampling_logp_difference/max": 1.9484190940856934, "sampling/importance_sampling_ratio/min": 0.142499178647995, "sampling/importance_sampling_ratio/mean": 0.9939104914665222, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24570434354245663, "clip_ratio/low_mean": 0.018391148187220097, "clip_ratio/low_min": 0.018391148187220097, "clip_ratio/high_mean": 0.05380117055028677, "clip_ratio/high_max": 0.05380117055028677, "clip_ratio/region_mean": 0.07219231873750687, "reward_total_mean": 0.840766429901123, "reward_meter_mean": 0.840766429901123, "reward_meter_std": 0.2658155858516693, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.840766429901123, "reward_total_composite_std": 0.2658155858516693, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1398.0} {"timestamp_utc": "2026-04-11T22:22:21Z", "mode": "train", "global_step": 1399, "epoch": 0.0540237874575224, "loss": -0.028, "grad_norm": 2.637542724609375, "learning_rate": 5.763636363636365e-06, "num_tokens": 3029544.0, "completions/mean_length": 233.0, "completions/min_length": 220.0, "completions/max_length": 253.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 233.0, "completions/min_terminated_length": 220.0, "completions/max_terminated_length": 253.0, "rewards/meter/mean": 0.4823773205280304, "rewards/meter/std": 0.34405437111854553, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.361782968044281, "rewards/total_composite/std": 0.25804078578948975, "reward": 0.361782968044281, "reward_std": 0.25804075598716736, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03455236554145813, "sampling/sampling_logp_difference/max": 4.196084022521973, "sampling/importance_sampling_ratio/min": 0.015054414980113506, "sampling/importance_sampling_ratio/mean": 1.0008431673049927, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08072605356574059, "clip_ratio/low_mean": 0.011501659755595028, "clip_ratio/low_min": 0.011501659755595028, "clip_ratio/high_mean": 0.0052334858337417245, "clip_ratio/high_max": 0.0052334858337417245, "clip_ratio/region_mean": 0.016735145589336753, "reward_total_mean": 0.361782968044281, "reward_meter_mean": 0.4823773205280304, "reward_meter_std": 0.34405437111854553, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.361782968044281, "reward_total_composite_std": 0.25804078578948975, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1399.0} {"timestamp_utc": "2026-04-11T22:22:27Z", "mode": "train", "global_step": 1400, "epoch": 0.05406240345999382, "loss": -0.0473, "grad_norm": 1.5132702589035034, "learning_rate": 5.760606060606061e-06, "num_tokens": 3032594.0, "completions/mean_length": 199.25, "completions/min_length": 175.0, "completions/max_length": 220.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 199.25, "completions/min_terminated_length": 175.0, "completions/max_terminated_length": 220.0, "rewards/meter/mean": 0.8474883437156677, "rewards/meter/std": 0.34002411365509033, "rewards/count_adherence/mean": 0.8392857313156128, "rewards/count_adherence/std": 0.05050762742757797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7260978817939758, "rewards/total_composite/std": 0.2923434376716614, "reward": 0.7260978817939758, "reward_std": 0.2923434376716614, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013634543865919113, "sampling/sampling_logp_difference/max": 1.0964598655700684, "sampling/importance_sampling_ratio/min": 0.33405157923698425, "sampling/importance_sampling_ratio/mean": 0.9997726678848267, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.057425904320552945, "clip_ratio/low_mean": 0.005651834304444492, "clip_ratio/low_min": 0.005651834304444492, "clip_ratio/high_mean": 0.008669710718095303, "clip_ratio/high_max": 0.008669710718095303, "clip_ratio/region_mean": 0.014321545022539794, "reward_total_mean": 0.7260978817939758, "reward_meter_mean": 0.8474883437156677, "reward_meter_std": 0.34002411365509033, "reward_count_adherence_mean": 0.8392857313156128, "reward_count_adherence_std": 0.05050762742757797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7260978817939758, "reward_total_composite_std": 0.2923434376716614, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1400.0} {"timestamp_utc": "2026-04-11T22:23:50Z", "mode": "eval", "global_step": 1400, "epoch": 0.05406240345999382, "eval_loss": NaN, "eval_runtime": 82.0154, "eval_samples_per_second": 1.268, "eval_steps_per_second": 0.159, "eval_num_tokens": 3032594.0, "eval_completions/mean_length": 208.76923076923077, "eval_completions/min_length": 59.23076923076923, "eval_completions/max_length": 442.84615384615387, "eval_completions/clipped_ratio": 0.0673076923076923, "eval_completions/mean_terminated_length": 187.3232644888071, "eval_completions/min_terminated_length": 59.23076923076923, "eval_completions/max_terminated_length": 375.0769230769231, "eval_rewards/meter/mean": 0.7005750903716454, "eval_rewards/meter/std": 0.3497193742256898, "eval_rewards/count_adherence/mean": 0.757807025542626, "eval_rewards/count_adherence/std": 0.20436857268214226, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/total_composite/mean": 0.5641365211743575, "eval_rewards/total_composite/std": 0.3036496341228485, "eval_reward": 0.5641365211743575, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0050584726357975835, "eval_sampling/sampling_logp_difference/max": 0.6686469912528992, "eval_sampling/importance_sampling_ratio/min": 0.5453932686493947, "eval_sampling/importance_sampling_ratio/mean": 1.0009340414634118, "eval_sampling/importance_sampling_ratio/max": 1.3240029720159678, "eval_entropy": 0.03806114311401661, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5641365211743575, "eval_reward_meter_mean": 0.7005750903716454, "eval_reward_meter_std": 0.3497193742256898, "eval_reward_count_adherence_mean": 0.757807025542626, "eval_reward_count_adherence_std": 0.20436857268214226, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_total_composite_mean": 0.5641365211743575, "eval_reward_total_composite_std": 0.3036496341228485, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1400.0} {"timestamp_utc": "2026-04-11T22:24:00Z", "mode": "train", "global_step": 1401, "epoch": 0.05410101946246525, "loss": -0.0038, "grad_norm": 1.0778833627700806, "learning_rate": 5.7575757575757586e-06, "num_tokens": 3034554.0, "completions/mean_length": 77.0, "completions/min_length": 76.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.0, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9939175844192505, "rewards/meter/std": 0.000148229009937495, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9939175844192505, "rewards/total_composite/std": 0.000148229009937495, "reward": 0.9939175844192505, "reward_std": 0.00014822981029283255, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009183553047478199, "sampling/sampling_logp_difference/max": 1.8037073612213135, "sampling/importance_sampling_ratio/min": 0.16468720138072968, "sampling/importance_sampling_ratio/mean": 0.9980778098106384, "sampling/importance_sampling_ratio/max": 1.225674033164978, "entropy": 0.023621314903721213, "clip_ratio/low_mean": 0.004912850330583751, "clip_ratio/low_min": 0.004912850330583751, "clip_ratio/high_mean": 0.004807692370377481, "clip_ratio/high_max": 0.004807692370377481, "clip_ratio/region_mean": 0.009720542700961232, "reward_total_mean": 0.9939175844192505, "reward_meter_mean": 0.9939175844192505, "reward_meter_std": 0.000148229009937495, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9939175844192505, "reward_total_composite_std": 0.000148229009937495, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1401.0} {"timestamp_utc": "2026-04-11T22:24:05Z", "mode": "train", "global_step": 1402, "epoch": 0.05413963546493667, "loss": -0.0589, "grad_norm": 0.4459538161754608, "learning_rate": 5.754545454545455e-06, "num_tokens": 3036262.0, "completions/mean_length": 62.5, "completions/min_length": 61.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9313734173774719, "rewards/meter/std": 0.006508991122245789, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6596269607543945, "rewards/total_composite/std": 0.10895697772502899, "reward": 0.6596269607543945, "reward_std": 0.1089569702744484, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005877028685063124, "sampling/sampling_logp_difference/max": 0.7637443542480469, "sampling/importance_sampling_ratio/min": 0.4659186005592346, "sampling/importance_sampling_ratio/mean": 0.9976125955581665, "sampling/importance_sampling_ratio/max": 1.1865988969802856, "entropy": 0.014842505101114511, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/high_mean": 0.0017123287543654442, "clip_ratio/high_max": 0.0017123287543654442, "clip_ratio/region_mean": 0.005810689181089401, "reward_total_mean": 0.6596269607543945, "reward_meter_mean": 0.9313734173774719, "reward_meter_std": 0.006508991122245789, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6596269607543945, "reward_total_composite_std": 0.10895697772502899, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1402.0} {"timestamp_utc": "2026-04-11T22:24:10Z", "mode": "train", "global_step": 1403, "epoch": 0.054178251467408095, "loss": 0.053, "grad_norm": 4.730922698974609, "learning_rate": 5.751515151515152e-06, "num_tokens": 3038160.0, "completions/mean_length": 90.25, "completions/min_length": 83.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.25, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.624970555305481, "rewards/meter/std": 0.3839372396469116, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.624970555305481, "rewards/total_composite/std": 0.3839372396469116, "reward": 0.624970555305481, "reward_std": 0.38393720984458923, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03412361070513725, "sampling/sampling_logp_difference/max": 2.2702224254608154, "sampling/importance_sampling_ratio/min": 0.10328920930624008, "sampling/importance_sampling_ratio/mean": 1.00449538230896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1069687600247562, "clip_ratio/low_mean": 0.010557568981312215, "clip_ratio/low_min": 0.010557568981312215, "clip_ratio/high_mean": 0.018244944512844086, "clip_ratio/high_max": 0.018244944512844086, "clip_ratio/region_mean": 0.0288025134941563, "reward_total_mean": 0.624970555305481, "reward_meter_mean": 0.624970555305481, "reward_meter_std": 0.3839372396469116, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.624970555305481, "reward_total_composite_std": 0.3839372396469116, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1403.0} {"timestamp_utc": "2026-04-11T22:24:15Z", "mode": "train", "global_step": 1404, "epoch": 0.05421686746987952, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.748484848484849e-06, "num_tokens": 3039888.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9974405169487, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9974405169487, "rewards/total_composite/std": 0.0, "reward": 0.9974405169487, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0008566752658225596, "sampling/sampling_logp_difference/max": 0.03150223195552826, "sampling/importance_sampling_ratio/min": 0.973335325717926, "sampling/importance_sampling_ratio/mean": 1.0007158517837524, "sampling/importance_sampling_ratio/max": 1.03200364112854, "entropy": 0.008136454911436886, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9974405169487, "reward_meter_mean": 0.9974405169487, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9974405169487, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1404.0} {"timestamp_utc": "2026-04-11T22:24:20Z", "mode": "train", "global_step": 1405, "epoch": 0.05425548347235094, "loss": -0.0621, "grad_norm": 8.912030220031738, "learning_rate": 5.745454545454546e-06, "num_tokens": 3041955.0, "completions/mean_length": 101.375, "completions/min_length": 88.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.375, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9784089922904968, "rewards/meter/std": 0.0068154833279550076, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8967825770378113, "rewards/total_composite/std": 0.15061092376708984, "reward": 0.8967825770378113, "reward_std": 0.15061092376708984, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013031724840402603, "sampling/sampling_logp_difference/max": 1.2515859603881836, "sampling/importance_sampling_ratio/min": 0.2860507667064667, "sampling/importance_sampling_ratio/mean": 1.000883936882019, "sampling/importance_sampling_ratio/max": 1.5073766708374023, "entropy": 0.05351943522691727, "clip_ratio/low_mean": 0.0014044943964108825, "clip_ratio/low_min": 0.0014044943964108825, "clip_ratio/high_mean": 0.007065693265758455, "clip_ratio/high_max": 0.007065693265758455, "clip_ratio/region_mean": 0.008470187662169337, "reward_total_mean": 0.8967825770378113, "reward_meter_mean": 0.9784089922904968, "reward_meter_std": 0.0068154833279550076, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8967825770378113, "reward_total_composite_std": 0.15061092376708984, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1405.0} {"timestamp_utc": "2026-04-11T22:24:26Z", "mode": "train", "global_step": 1406, "epoch": 0.05429409947482237, "loss": -0.0295, "grad_norm": 8.725440979003906, "learning_rate": 5.742424242424242e-06, "num_tokens": 3044096.0, "completions/mean_length": 106.625, "completions/min_length": 96.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.625, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.9755566120147705, "rewards/meter/std": 0.00819630827754736, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9351888298988342, "rewards/total_composite/std": 0.11715210229158401, "reward": 0.9351888298988342, "reward_std": 0.11715211719274521, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008083357475697994, "sampling/sampling_logp_difference/max": 1.5227665901184082, "sampling/importance_sampling_ratio/min": 0.21810764074325562, "sampling/importance_sampling_ratio/mean": 1.0007258653640747, "sampling/importance_sampling_ratio/max": 1.5461537837982178, "entropy": 0.03398331138305366, "clip_ratio/low_mean": 0.0013020833721384406, "clip_ratio/low_min": 0.0013020833721384406, "clip_ratio/high_mean": 0.004636763245798647, "clip_ratio/high_max": 0.004636763245798647, "clip_ratio/region_mean": 0.005938846617937088, "reward_total_mean": 0.9351888298988342, "reward_meter_mean": 0.9755566120147705, "reward_meter_std": 0.00819630827754736, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9351888298988342, "reward_total_composite_std": 0.11715210229158401, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1406.0} {"timestamp_utc": "2026-04-11T22:24:32Z", "mode": "train", "global_step": 1407, "epoch": 0.05433271547729379, "loss": 0.0654, "grad_norm": 9.687000274658203, "learning_rate": 5.73939393939394e-06, "num_tokens": 3046947.0, "completions/mean_length": 160.375, "completions/min_length": 141.0, "completions/max_length": 195.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.375, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 195.0, "rewards/meter/mean": 0.5095553994178772, "rewards/meter/std": 0.306953102350235, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4246295094490051, "rewards/total_composite/std": 0.2557942271232605, "reward": 0.4246295094490051, "reward_std": 0.2557942271232605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030556051060557365, "sampling/sampling_logp_difference/max": 2.4804577827453613, "sampling/importance_sampling_ratio/min": 0.08370490372180939, "sampling/importance_sampling_ratio/mean": 1.0008848905563354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1448420793749392, "clip_ratio/low_mean": 0.017260977067053318, "clip_ratio/low_min": 0.017260977067053318, "clip_ratio/high_mean": 0.004213172011077404, "clip_ratio/high_max": 0.004213172011077404, "clip_ratio/region_mean": 0.021474149078130722, "reward_total_mean": 0.4246295094490051, "reward_meter_mean": 0.5095553994178772, "reward_meter_std": 0.306953102350235, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4246295094490051, "reward_total_composite_std": 0.2557942271232605, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1407.0} {"timestamp_utc": "2026-04-11T22:24:38Z", "mode": "train", "global_step": 1408, "epoch": 0.054371331479765216, "loss": 0.0424, "grad_norm": 3.7705280780792236, "learning_rate": 5.736363636363637e-06, "num_tokens": 3049015.0, "completions/mean_length": 95.5, "completions/min_length": 90.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.5, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.8144540190696716, "rewards/meter/std": 0.3588285744190216, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8144540190696716, "rewards/total_composite/std": 0.3588285744190216, "reward": 0.8144540190696716, "reward_std": 0.3588285744190216, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03320210054516792, "sampling/sampling_logp_difference/max": 1.875186800956726, "sampling/importance_sampling_ratio/min": 0.15332631766796112, "sampling/importance_sampling_ratio/mean": 0.9958533048629761, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11370669398456812, "clip_ratio/low_mean": 0.008796296548098326, "clip_ratio/low_min": 0.008796296548098326, "clip_ratio/high_mean": 0.02524956443812698, "clip_ratio/high_max": 0.02524956443812698, "clip_ratio/region_mean": 0.03404586098622531, "reward_total_mean": 0.8144540190696716, "reward_meter_mean": 0.8144540190696716, "reward_meter_std": 0.3588285744190216, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8144540190696716, "reward_total_composite_std": 0.3588285744190216, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1408.0} {"timestamp_utc": "2026-04-11T22:24:42Z", "mode": "train", "global_step": 1409, "epoch": 0.05440994748223664, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.733333333333334e-06, "num_tokens": 3050751.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9974405169487, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9974405169487, "rewards/total_composite/std": 0.0, "reward": 0.9974405169487, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006318983505479991, "sampling/sampling_logp_difference/max": 0.026535823941230774, "sampling/importance_sampling_ratio/min": 0.9997444152832031, "sampling/importance_sampling_ratio/mean": 1.000633716583252, "sampling/importance_sampling_ratio/max": 1.0268909931182861, "entropy": 0.005596369504928589, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9974405169487, "reward_meter_mean": 0.9974405169487, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9974405169487, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1409.0} {"timestamp_utc": "2026-04-11T22:24:47Z", "mode": "train", "global_step": 1410, "epoch": 0.054448563484708064, "loss": -0.0006, "grad_norm": 0.3544546365737915, "learning_rate": 5.7303030303030305e-06, "num_tokens": 3052484.0, "completions/mean_length": 51.625, "completions/min_length": 51.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9937834739685059, "rewards/meter/std": 1.4745512999070343e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9937834739685059, "rewards/total_composite/std": 1.4745512999070343e-05, "reward": 0.9937834739685059, "reward_std": 1.4749624824617058e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010635309852659702, "sampling/sampling_logp_difference/max": 1.267793893814087, "sampling/importance_sampling_ratio/min": 0.28145185112953186, "sampling/importance_sampling_ratio/mean": 0.9973414540290833, "sampling/importance_sampling_ratio/max": 1.660771131515503, "entropy": 0.009844112151768059, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.009615384973585606, "clip_ratio/high_max": 0.009615384973585606, "clip_ratio/region_mean": 0.009615384973585606, "reward_total_mean": 0.9937834739685059, "reward_meter_mean": 0.9937834739685059, "reward_meter_std": 1.4745512999070343e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9937834739685059, "reward_total_composite_std": 1.4745512999070343e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1410.0} {"timestamp_utc": "2026-04-11T22:24:54Z", "mode": "train", "global_step": 1411, "epoch": 0.05448717948717949, "loss": 0.0372, "grad_norm": 10.747732162475586, "learning_rate": 5.727272727272728e-06, "num_tokens": 3055704.0, "completions/mean_length": 199.5, "completions/min_length": 182.0, "completions/max_length": 223.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 199.5, "completions/min_terminated_length": 182.0, "completions/max_terminated_length": 223.0, "rewards/meter/mean": 0.917418360710144, "rewards/meter/std": 0.19984416663646698, "rewards/count_adherence/mean": 0.828125, "rewards/count_adherence/std": 0.06469365209341049, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7559784650802612, "rewards/total_composite/std": 0.1650553196668625, "reward": 0.7559784650802612, "reward_std": 0.1650553196668625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021405506879091263, "sampling/sampling_logp_difference/max": 2.6951780319213867, "sampling/importance_sampling_ratio/min": 0.06753035634756088, "sampling/importance_sampling_ratio/mean": 0.9983882904052734, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07436814438551664, "clip_ratio/low_mean": 0.009086228616070002, "clip_ratio/low_min": 0.009086228616070002, "clip_ratio/high_mean": 0.005750448792241514, "clip_ratio/high_max": 0.005750448792241514, "clip_ratio/region_mean": 0.014836677408311516, "reward_total_mean": 0.7559784650802612, "reward_meter_mean": 0.917418360710144, "reward_meter_std": 0.19984416663646698, "reward_count_adherence_mean": 0.828125, "reward_count_adherence_std": 0.06469365209341049, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7559784650802612, "reward_total_composite_std": 0.1650553196668625, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1411.0} {"timestamp_utc": "2026-04-11T22:25:00Z", "mode": "train", "global_step": 1412, "epoch": 0.05452579548965091, "loss": -0.0047, "grad_norm": 3.045875072479248, "learning_rate": 5.724242424242424e-06, "num_tokens": 3058447.0, "completions/mean_length": 157.875, "completions/min_length": 142.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.875, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.9747276306152344, "rewards/meter/std": 0.03132884204387665, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9747276306152344, "rewards/total_composite/std": 0.03132884204387665, "reward": 0.9747276306152344, "reward_std": 0.03132886067032814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03142685815691948, "sampling/sampling_logp_difference/max": 1.717909336090088, "sampling/importance_sampling_ratio/min": 0.1794409155845642, "sampling/importance_sampling_ratio/mean": 0.9941987991333008, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10427726805210114, "clip_ratio/low_mean": 0.012016088468953967, "clip_ratio/low_min": 0.012016088468953967, "clip_ratio/high_mean": 0.013903069775551558, "clip_ratio/high_max": 0.013903069775551558, "clip_ratio/region_mean": 0.025919158244505525, "reward_total_mean": 0.9747276306152344, "reward_meter_mean": 0.9747276306152344, "reward_meter_std": 0.03132884204387665, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9747276306152344, "reward_total_composite_std": 0.03132884204387665, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1412.0} {"timestamp_utc": "2026-04-11T22:25:04Z", "mode": "train", "global_step": 1413, "epoch": 0.054564411492122336, "loss": 0.0287, "grad_norm": 8.96839714050293, "learning_rate": 5.721212121212122e-06, "num_tokens": 3060005.0, "completions/mean_length": 35.75, "completions/min_length": 33.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.21988053619861603, "rewards/meter/std": 0.18961991369724274, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.21988053619861603, "rewards/total_composite/std": 0.18961991369724274, "reward": 0.21988053619861603, "reward_std": 0.18961989879608154, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029288217425346375, "sampling/sampling_logp_difference/max": 0.9800626039505005, "sampling/importance_sampling_ratio/min": 0.37528759241104126, "sampling/importance_sampling_ratio/mean": 0.9913184642791748, "sampling/importance_sampling_ratio/max": 1.9583380222320557, "entropy": 0.08843746781349182, "clip_ratio/low_mean": 0.010416666744276881, "clip_ratio/low_min": 0.010416666744276881, "clip_ratio/high_mean": 0.021148990374058485, "clip_ratio/high_max": 0.021148990374058485, "clip_ratio/region_mean": 0.031565657118335366, "reward_total_mean": 0.21988053619861603, "reward_meter_mean": 0.21988053619861603, "reward_meter_std": 0.18961991369724274, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.21988053619861603, "reward_total_composite_std": 0.18961991369724274, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1413.0} {"timestamp_utc": "2026-04-11T22:25:12Z", "mode": "train", "global_step": 1414, "epoch": 0.05460302749459376, "loss": 0.0229, "grad_norm": 2.493736743927002, "learning_rate": 5.718181818181819e-06, "num_tokens": 3064258.0, "completions/mean_length": 338.625, "completions/min_length": 320.0, "completions/max_length": 361.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 338.625, "completions/min_terminated_length": 320.0, "completions/max_terminated_length": 361.0, "rewards/meter/mean": 0.9436893463134766, "rewards/meter/std": 0.14909327030181885, "rewards/count_adherence/mean": 0.7946428060531616, "rewards/count_adherence/std": 0.025253823027014732, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7503798007965088, "rewards/total_composite/std": 0.12338951230049133, "reward": 0.7503798007965088, "reward_std": 0.12338951230049133, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014010339975357056, "sampling/sampling_logp_difference/max": 2.0716729164123535, "sampling/importance_sampling_ratio/min": 0.1259748637676239, "sampling/importance_sampling_ratio/mean": 0.999437153339386, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0529011411126703, "clip_ratio/low_mean": 0.0006944444612599909, "clip_ratio/low_min": 0.0006944444612599909, "clip_ratio/high_mean": 0.009912769484799355, "clip_ratio/high_max": 0.009912769484799355, "clip_ratio/region_mean": 0.010607213946059346, "reward_total_mean": 0.7503798007965088, "reward_meter_mean": 0.9436893463134766, "reward_meter_std": 0.14909327030181885, "reward_count_adherence_mean": 0.7946428060531616, "reward_count_adherence_std": 0.025253823027014732, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7503798007965088, "reward_total_composite_std": 0.12338951230049133, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1414.0} {"timestamp_utc": "2026-04-11T22:25:18Z", "mode": "train", "global_step": 1415, "epoch": 0.054641643497065184, "loss": 0.0262, "grad_norm": 15.694778442382812, "learning_rate": 5.715151515151516e-06, "num_tokens": 3066737.0, "completions/mean_length": 149.875, "completions/min_length": 132.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.875, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.8777439594268799, "rewards/meter/std": 0.2111976444721222, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8777439594268799, "rewards/total_composite/std": 0.2111976444721222, "reward": 0.8777439594268799, "reward_std": 0.2111976593732834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008209249004721642, "sampling/sampling_logp_difference/max": 1.9362608194351196, "sampling/importance_sampling_ratio/min": 0.1442423015832901, "sampling/importance_sampling_ratio/mean": 1.0013052225112915, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.022122560651041567, "clip_ratio/low_mean": 0.002419354859739542, "clip_ratio/low_min": 0.002419354859739542, "clip_ratio/high_mean": 0.0050054112216457725, "clip_ratio/high_max": 0.0050054112216457725, "clip_ratio/region_mean": 0.0074247660813853145, "reward_total_mean": 0.8777439594268799, "reward_meter_mean": 0.8777439594268799, "reward_meter_std": 0.2111976444721222, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8777439594268799, "reward_total_composite_std": 0.2111976444721222, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1415.0} {"timestamp_utc": "2026-04-11T22:25:25Z", "mode": "train", "global_step": 1416, "epoch": 0.05468025949953661, "loss": 0.0148, "grad_norm": 1.8256059885025024, "learning_rate": 5.712121212121212e-06, "num_tokens": 3069742.0, "completions/mean_length": 189.625, "completions/min_length": 183.0, "completions/max_length": 191.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 189.625, "completions/min_terminated_length": 183.0, "completions/max_terminated_length": 191.0, "rewards/meter/mean": 0.9700912237167358, "rewards/meter/std": 0.0033568073995411396, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7275683879852295, "rewards/total_composite/std": 0.00251759379170835, "reward": 0.7275683879852295, "reward_std": 0.002517581917345524, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006129469722509384, "sampling/sampling_logp_difference/max": 1.342543125152588, "sampling/importance_sampling_ratio/min": 0.2611806094646454, "sampling/importance_sampling_ratio/mean": 1.0012221336364746, "sampling/importance_sampling_ratio/max": 1.9661462306976318, "entropy": 0.028316745534539223, "clip_ratio/low_mean": 0.000654450268484652, "clip_ratio/low_min": 0.000654450268484652, "clip_ratio/high_mean": 0.004043861059471965, "clip_ratio/high_max": 0.004043861059471965, "clip_ratio/region_mean": 0.004698311327956617, "reward_total_mean": 0.7275683879852295, "reward_meter_mean": 0.9700912237167358, "reward_meter_std": 0.0033568073995411396, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7275683879852295, "reward_total_composite_std": 0.00251759379170835, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1416.0} {"timestamp_utc": "2026-04-11T22:25:30Z", "mode": "train", "global_step": 1417, "epoch": 0.05471887550200803, "loss": 0.066, "grad_norm": 10.990321159362793, "learning_rate": 5.7090909090909096e-06, "num_tokens": 3071620.0, "completions/mean_length": 72.75, "completions/min_length": 68.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.75, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.8147554397583008, "rewards/meter/std": 0.2788236141204834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8147554397583008, "rewards/total_composite/std": 0.2788236141204834, "reward": 0.8147554397583008, "reward_std": 0.2788236439228058, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019177932292222977, "sampling/sampling_logp_difference/max": 1.5638766288757324, "sampling/importance_sampling_ratio/min": 0.20932303369045258, "sampling/importance_sampling_ratio/mean": 1.0003993511199951, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07261863665189594, "clip_ratio/low_mean": 0.007640770636498928, "clip_ratio/low_min": 0.007640770636498928, "clip_ratio/high_mean": 0.0069076798390597105, "clip_ratio/high_max": 0.0069076798390597105, "clip_ratio/region_mean": 0.014548450475558639, "reward_total_mean": 0.8147554397583008, "reward_meter_mean": 0.8147554397583008, "reward_meter_std": 0.2788236141204834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8147554397583008, "reward_total_composite_std": 0.2788236141204834, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1417.0} {"timestamp_utc": "2026-04-11T22:25:39Z", "mode": "train", "global_step": 1418, "epoch": 0.05475749150447946, "loss": 0.0409, "grad_norm": 0.818814218044281, "learning_rate": 5.706060606060606e-06, "num_tokens": 3077089.0, "completions/mean_length": 454.625, "completions/min_length": 418.0, "completions/max_length": 478.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 454.625, "completions/min_terminated_length": 418.0, "completions/max_terminated_length": 478.0, "rewards/meter/mean": 0.9967572689056396, "rewards/meter/std": 0.006844592746347189, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.033064987510442734, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8722063302993774, "rewards/total_composite/std": 0.03481004759669304, "reward": 0.8722063302993774, "reward_std": 0.03481004387140274, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006249036639928818, "sampling/sampling_logp_difference/max": 3.112579107284546, "sampling/importance_sampling_ratio/min": 0.04448607563972473, "sampling/importance_sampling_ratio/mean": 0.999168872833252, "sampling/importance_sampling_ratio/max": 1.7395983934402466, "entropy": 0.019142753910273314, "clip_ratio/low_mean": 0.0038076729979366064, "clip_ratio/low_min": 0.0038076729979366064, "clip_ratio/high_mean": 0.001791403570678085, "clip_ratio/high_max": 0.001791403570678085, "clip_ratio/region_mean": 0.0055990765686146915, "reward_total_mean": 0.8722063302993774, "reward_meter_mean": 0.9967572689056396, "reward_meter_std": 0.006844592746347189, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.033064987510442734, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8722063302993774, "reward_total_composite_std": 0.03481004759669304, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1418.0} {"timestamp_utc": "2026-04-11T22:25:44Z", "mode": "train", "global_step": 1419, "epoch": 0.05479610750695088, "loss": 0.0253, "grad_norm": 6.161533832550049, "learning_rate": 5.703030303030303e-06, "num_tokens": 3079007.0, "completions/mean_length": 60.75, "completions/min_length": 58.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9039314985275269, "rewards/meter/std": 0.130006805062294, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9039314985275269, "rewards/total_composite/std": 0.130006805062294, "reward": 0.9039314985275269, "reward_std": 0.1300068199634552, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02613363415002823, "sampling/sampling_logp_difference/max": 0.9707651138305664, "sampling/importance_sampling_ratio/min": 0.3787930905818939, "sampling/importance_sampling_ratio/mean": 0.9989476203918457, "sampling/importance_sampling_ratio/max": 1.948388695716858, "entropy": 0.1282052816823125, "clip_ratio/low_mean": 0.005989583441987634, "clip_ratio/low_min": 0.005989583441987634, "clip_ratio/high_mean": 0.028853219700977206, "clip_ratio/high_max": 0.028853219700977206, "clip_ratio/region_mean": 0.03484280314296484, "reward_total_mean": 0.9039314985275269, "reward_meter_mean": 0.9039314985275269, "reward_meter_std": 0.130006805062294, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9039314985275269, "reward_total_composite_std": 0.130006805062294, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1419.0} {"timestamp_utc": "2026-04-11T22:25:49Z", "mode": "train", "global_step": 1420, "epoch": 0.054834723509422305, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.7e-06, "num_tokens": 3080967.0, "completions/mean_length": 65.0, "completions/min_length": 65.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9937941431999207, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9937941431999207, "rewards/total_composite/std": 0.0, "reward": 0.9937941431999207, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 5.171665543457493e-05, "sampling/sampling_logp_difference/max": 0.0018144652713090181, "sampling/importance_sampling_ratio/min": 0.998187243938446, "sampling/importance_sampling_ratio/mean": 1.0000417232513428, "sampling/importance_sampling_ratio/max": 1.001667857170105, "entropy": 0.0005712458550988231, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9937941431999207, "reward_meter_mean": 0.9937941431999207, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9937941431999207, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1420.0} {"timestamp_utc": "2026-04-11T22:25:55Z", "mode": "train", "global_step": 1421, "epoch": 0.05487333951189373, "loss": -0.0073, "grad_norm": 3.501631021499634, "learning_rate": 5.696969696969698e-06, "num_tokens": 3084586.0, "completions/mean_length": 220.375, "completions/min_length": 206.0, "completions/max_length": 234.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 220.375, "completions/min_terminated_length": 206.0, "completions/max_terminated_length": 234.0, "rewards/meter/mean": 0.8429533243179321, "rewards/meter/std": 0.2051660567522049, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8429533243179321, "rewards/total_composite/std": 0.2051660567522049, "reward": 0.8429533243179321, "reward_std": 0.2051660418510437, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03215697035193443, "sampling/sampling_logp_difference/max": 5.8938775062561035, "sampling/importance_sampling_ratio/min": 0.0027562687173485756, "sampling/importance_sampling_ratio/mean": 0.9992908835411072, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08196912333369255, "clip_ratio/low_mean": 0.0076804916607216, "clip_ratio/low_min": 0.0076804916607216, "clip_ratio/high_mean": 0.014449202571995556, "clip_ratio/high_max": 0.014449202571995556, "clip_ratio/region_mean": 0.022129694232717156, "reward_total_mean": 0.8429533243179321, "reward_meter_mean": 0.8429533243179321, "reward_meter_std": 0.2051660567522049, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8429533243179321, "reward_total_composite_std": 0.2051660567522049, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1421.0} {"timestamp_utc": "2026-04-11T22:26:00Z", "mode": "train", "global_step": 1422, "epoch": 0.05491195551436515, "loss": 0.0732, "grad_norm": 9.643394470214844, "learning_rate": 5.693939393939394e-06, "num_tokens": 3086116.0, "completions/mean_length": 56.25, "completions/min_length": 47.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5467125177383423, "rewards/meter/std": 0.30351436138153076, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5467125177383423, "rewards/total_composite/std": 0.30351436138153076, "reward": 0.5467125177383423, "reward_std": 0.3035143315792084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04777029901742935, "sampling/sampling_logp_difference/max": 1.6549041271209717, "sampling/importance_sampling_ratio/min": 0.19111038744449615, "sampling/importance_sampling_ratio/mean": 1.0036096572875977, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11466897837817669, "clip_ratio/low_mean": 0.02890711883082986, "clip_ratio/low_min": 0.02890711883082986, "clip_ratio/high_mean": 0.01656103180721402, "clip_ratio/high_max": 0.01656103180721402, "clip_ratio/region_mean": 0.04546815063804388, "reward_total_mean": 0.5467125177383423, "reward_meter_mean": 0.5467125177383423, "reward_meter_std": 0.30351436138153076, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5467125177383423, "reward_total_composite_std": 0.30351436138153076, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1422.0} {"timestamp_utc": "2026-04-11T22:26:06Z", "mode": "train", "global_step": 1423, "epoch": 0.05495057151683658, "loss": 0.0134, "grad_norm": 3.3753459453582764, "learning_rate": 5.690909090909091e-06, "num_tokens": 3088707.0, "completions/mean_length": 153.875, "completions/min_length": 148.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 153.875, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.9830798506736755, "rewards/meter/std": 0.006752775516360998, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7373098731040955, "rewards/total_composite/std": 0.0050645796582102776, "reward": 0.7373098731040955, "reward_std": 0.005064563360065222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013763805851340294, "sampling/sampling_logp_difference/max": 1.1305841207504272, "sampling/importance_sampling_ratio/min": 0.32284462451934814, "sampling/importance_sampling_ratio/mean": 0.9999931454658508, "sampling/importance_sampling_ratio/max": 1.6671247482299805, "entropy": 0.04576628375798464, "clip_ratio/low_mean": 0.0016025641234591603, "clip_ratio/low_min": 0.0016025641234591603, "clip_ratio/high_mean": 0.00249047129182145, "clip_ratio/high_max": 0.00249047129182145, "clip_ratio/region_mean": 0.00409303541528061, "reward_total_mean": 0.7373098731040955, "reward_meter_mean": 0.9830798506736755, "reward_meter_std": 0.006752775516360998, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7373098731040955, "reward_total_composite_std": 0.0050645796582102776, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1423.0} {"timestamp_utc": "2026-04-11T22:26:11Z", "mode": "train", "global_step": 1424, "epoch": 0.054989187519308, "loss": 0.0012, "grad_norm": 2.6433005332946777, "learning_rate": 5.687878787878789e-06, "num_tokens": 3090899.0, "completions/mean_length": 97.0, "completions/min_length": 97.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.0, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9049856662750244, "rewards/meter/std": 0.009677063673734665, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.603323757648468, "rewards/total_composite/std": 0.006451376248151064, "reward": 0.603323757648468, "reward_std": 0.006451369728893042, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0032425601966679096, "sampling/sampling_logp_difference/max": 0.4332277774810791, "sampling/importance_sampling_ratio/min": 0.6484127640724182, "sampling/importance_sampling_ratio/mean": 0.9982846975326538, "sampling/importance_sampling_ratio/max": 1.138208270072937, "entropy": 0.011988057813141495, "clip_ratio/low_mean": 0.002577319508418441, "clip_ratio/low_min": 0.002577319508418441, "clip_ratio/high_mean": 0.002577319508418441, "clip_ratio/high_max": 0.002577319508418441, "clip_ratio/region_mean": 0.005154639016836882, "reward_total_mean": 0.603323757648468, "reward_meter_mean": 0.9049856662750244, "reward_meter_std": 0.009677063673734665, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.603323757648468, "reward_total_composite_std": 0.006451376248151064, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1424.0} {"timestamp_utc": "2026-04-11T22:26:17Z", "mode": "train", "global_step": 1425, "epoch": 0.055027803521779425, "loss": 0.0277, "grad_norm": 3.900181531906128, "learning_rate": 5.684848484848485e-06, "num_tokens": 3093822.0, "completions/mean_length": 183.375, "completions/min_length": 165.0, "completions/max_length": 196.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 183.375, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 196.0, "rewards/meter/mean": 0.7883792519569397, "rewards/meter/std": 0.30727648735046387, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6307033896446228, "rewards/total_composite/std": 0.24582119286060333, "reward": 0.6307033896446228, "reward_std": 0.24582119286060333, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04060335457324982, "sampling/sampling_logp_difference/max": 1.9158062934875488, "sampling/importance_sampling_ratio/min": 0.1472230851650238, "sampling/importance_sampling_ratio/mean": 1.0001552104949951, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2714184243232012, "clip_ratio/low_mean": 0.005951206665486097, "clip_ratio/low_min": 0.005951206665486097, "clip_ratio/high_mean": 0.01706316671334207, "clip_ratio/high_max": 0.01706316671334207, "clip_ratio/region_mean": 0.023014373378828168, "reward_total_mean": 0.6307033896446228, "reward_meter_mean": 0.7883792519569397, "reward_meter_std": 0.30727648735046387, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6307033896446228, "reward_total_composite_std": 0.24582119286060333, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1425.0} {"timestamp_utc": "2026-04-11T22:26:25Z", "mode": "train", "global_step": 1426, "epoch": 0.05506641952425085, "loss": 0.0296, "grad_norm": 2.7521774768829346, "learning_rate": 5.681818181818183e-06, "num_tokens": 3097571.0, "completions/mean_length": 265.625, "completions/min_length": 258.0, "completions/max_length": 284.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 265.625, "completions/min_terminated_length": 258.0, "completions/max_terminated_length": 284.0, "rewards/meter/mean": 0.9851486682891846, "rewards/meter/std": 0.03613918274641037, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.844413161277771, "rewards/total_composite/std": 0.030976444482803345, "reward": 0.844413161277771, "reward_std": 0.030976444482803345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02135475166141987, "sampling/sampling_logp_difference/max": 5.254199028015137, "sampling/importance_sampling_ratio/min": 0.005225530359894037, "sampling/importance_sampling_ratio/mean": 0.9972927570343018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04411743604578078, "clip_ratio/low_mean": 0.0013204225106164813, "clip_ratio/low_min": 0.0013204225106164813, "clip_ratio/high_mean": 0.012304069881793112, "clip_ratio/high_max": 0.012304069881793112, "clip_ratio/region_mean": 0.013624492392409593, "reward_total_mean": 0.844413161277771, "reward_meter_mean": 0.9851486682891846, "reward_meter_std": 0.03613918274641037, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.844413161277771, "reward_total_composite_std": 0.030976444482803345, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1426.0} {"timestamp_utc": "2026-04-11T22:26:29Z", "mode": "train", "global_step": 1427, "epoch": 0.055105035526722274, "loss": -0.0243, "grad_norm": 10.523006439208984, "learning_rate": 5.67878787878788e-06, "num_tokens": 3099150.0, "completions/mean_length": 35.375, "completions/min_length": 32.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.8689415454864502, "rewards/meter/std": 0.32687532901763916, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8689415454864502, "rewards/total_composite/std": 0.32687532901763916, "reward": 0.8689415454864502, "reward_std": 0.32687532901763916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026072131469845772, "sampling/sampling_logp_difference/max": 1.0885493755340576, "sampling/importance_sampling_ratio/min": 0.3367045819759369, "sampling/importance_sampling_ratio/mean": 0.9974402189254761, "sampling/importance_sampling_ratio/max": 1.5091142654418945, "entropy": 0.11127344332635403, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.014055822743102908, "clip_ratio/high_max": 0.014055822743102908, "clip_ratio/region_mean": 0.014055822743102908, "reward_total_mean": 0.8689415454864502, "reward_meter_mean": 0.8689415454864502, "reward_meter_std": 0.32687532901763916, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8689415454864502, "reward_total_composite_std": 0.32687532901763916, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1427.0} {"timestamp_utc": "2026-04-11T22:26:36Z", "mode": "train", "global_step": 1428, "epoch": 0.0551436515291937, "loss": 0.0049, "grad_norm": 2.8291854858398438, "learning_rate": 5.675757575757577e-06, "num_tokens": 3102030.0, "completions/mean_length": 185.0, "completions/min_length": 171.0, "completions/max_length": 203.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 185.0, "completions/min_terminated_length": 171.0, "completions/max_terminated_length": 203.0, "rewards/meter/mean": 0.9146131277084351, "rewards/meter/std": 0.18409597873687744, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.731690526008606, "rewards/total_composite/std": 0.14727678894996643, "reward": 0.731690526008606, "reward_std": 0.14727677404880524, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028948500752449036, "sampling/sampling_logp_difference/max": 1.339238166809082, "sampling/importance_sampling_ratio/min": 0.2620452344417572, "sampling/importance_sampling_ratio/mean": 1.0074141025543213, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23660685354843736, "clip_ratio/low_mean": 0.002673796843737364, "clip_ratio/low_min": 0.002673796843737364, "clip_ratio/high_mean": 0.02136983221862465, "clip_ratio/high_max": 0.02136983221862465, "clip_ratio/region_mean": 0.024043629062362015, "reward_total_mean": 0.731690526008606, "reward_meter_mean": 0.9146131277084351, "reward_meter_std": 0.18409597873687744, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.731690526008606, "reward_total_composite_std": 0.14727678894996643, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1428.0} {"timestamp_utc": "2026-04-11T22:26:42Z", "mode": "train", "global_step": 1429, "epoch": 0.05518226753166512, "loss": 0.0126, "grad_norm": 2.0166499614715576, "learning_rate": 5.672727272727273e-06, "num_tokens": 3105352.0, "completions/mean_length": 217.25, "completions/min_length": 208.0, "completions/max_length": 224.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 217.25, "completions/min_terminated_length": 208.0, "completions/max_terminated_length": 224.0, "rewards/meter/mean": 0.7290028929710388, "rewards/meter/std": 0.27879202365875244, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7290028929710388, "rewards/total_composite/std": 0.27879202365875244, "reward": 0.7290028929710388, "reward_std": 0.27879202365875244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017551273107528687, "sampling/sampling_logp_difference/max": 2.811058282852173, "sampling/importance_sampling_ratio/min": 0.06014131382107735, "sampling/importance_sampling_ratio/mean": 1.0034372806549072, "sampling/importance_sampling_ratio/max": 1.7777851819992065, "entropy": 0.07306595658883452, "clip_ratio/low_mean": 0.004536679538432509, "clip_ratio/low_min": 0.004536679538432509, "clip_ratio/high_mean": 0.008716399548575282, "clip_ratio/high_max": 0.008716399548575282, "clip_ratio/region_mean": 0.01325307908700779, "reward_total_mean": 0.7290028929710388, "reward_meter_mean": 0.7290028929710388, "reward_meter_std": 0.27879202365875244, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7290028929710388, "reward_total_composite_std": 0.27879202365875244, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1429.0} {"timestamp_utc": "2026-04-11T22:26:49Z", "mode": "train", "global_step": 1430, "epoch": 0.055220883534136546, "loss": -0.0192, "grad_norm": 5.332306861877441, "learning_rate": 5.6696969696969705e-06, "num_tokens": 3107622.0, "completions/mean_length": 120.75, "completions/min_length": 105.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.75, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9565221071243286, "rewards/meter/std": 0.06291621178388596, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6376813650131226, "rewards/total_composite/std": 0.04194413870573044, "reward": 0.6376813650131226, "reward_std": 0.041944146156311035, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0366525761783123, "sampling/sampling_logp_difference/max": 2.4015703201293945, "sampling/importance_sampling_ratio/min": 0.09057561308145523, "sampling/importance_sampling_ratio/mean": 1.0100973844528198, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2136777131818235, "clip_ratio/low_mean": 0.014915966894477606, "clip_ratio/low_min": 0.014915966894477606, "clip_ratio/high_mean": 0.01930890593212098, "clip_ratio/high_max": 0.01930890593212098, "clip_ratio/region_mean": 0.034224872826598585, "reward_total_mean": 0.6376813650131226, "reward_meter_mean": 0.9565221071243286, "reward_meter_std": 0.06291621178388596, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6376813650131226, "reward_total_composite_std": 0.04194413870573044, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1430.0} {"timestamp_utc": "2026-04-11T22:26:54Z", "mode": "train", "global_step": 1431, "epoch": 0.05525949953660797, "loss": -0.0003, "grad_norm": 0.017195936292409897, "learning_rate": 5.666666666666667e-06, "num_tokens": 3109874.0, "completions/mean_length": 120.5, "completions/min_length": 117.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.5, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9971237182617188, "rewards/meter/std": 6.533247187689994e-07, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6647491455078125, "rewards/total_composite/std": 4.431865647802624e-07, "reward": 0.6647491455078125, "reward_std": 4.457557452042238e-07, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0020029370207339525, "sampling/sampling_logp_difference/max": 0.6329095363616943, "sampling/importance_sampling_ratio/min": 0.5310444831848145, "sampling/importance_sampling_ratio/mean": 1.0000869035720825, "sampling/importance_sampling_ratio/max": 1.200929880142212, "entropy": 0.009257287660147995, "clip_ratio/low_mean": 0.0010683761211112142, "clip_ratio/low_min": 0.0010683761211112142, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0010683761211112142, "reward_total_mean": 0.6647491455078125, "reward_meter_mean": 0.9971237182617188, "reward_meter_std": 6.533247187689994e-07, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6647491455078125, "reward_total_composite_std": 4.431865647802624e-07, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1431.0} {"timestamp_utc": "2026-04-11T22:26:59Z", "mode": "train", "global_step": 1432, "epoch": 0.055298115539079394, "loss": 0.0101, "grad_norm": 3.690661668777466, "learning_rate": 5.663636363636364e-06, "num_tokens": 3111886.0, "completions/mean_length": 75.5, "completions/min_length": 74.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.5, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9783248901367188, "rewards/meter/std": 0.05031197890639305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9783248901367188, "rewards/total_composite/std": 0.05031197890639305, "reward": 0.9783248901367188, "reward_std": 0.050311967730522156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040666528046131134, "sampling/sampling_logp_difference/max": 1.5964641571044922, "sampling/importance_sampling_ratio/min": 0.20261165499687195, "sampling/importance_sampling_ratio/mean": 0.990241527557373, "sampling/importance_sampling_ratio/max": 1.87123441696167, "entropy": 0.13668694626539946, "clip_ratio/low_mean": 0.00657894741743803, "clip_ratio/low_min": 0.00657894741743803, "clip_ratio/high_mean": 0.039813138311728835, "clip_ratio/high_max": 0.039813138311728835, "clip_ratio/region_mean": 0.046392085729166865, "reward_total_mean": 0.9783248901367188, "reward_meter_mean": 0.9783248901367188, "reward_meter_std": 0.05031197890639305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9783248901367188, "reward_total_composite_std": 0.05031197890639305, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1432.0} {"timestamp_utc": "2026-04-11T22:27:06Z", "mode": "train", "global_step": 1433, "epoch": 0.05533673154155082, "loss": 0.0156, "grad_norm": 2.917558431625366, "learning_rate": 5.6606060606060606e-06, "num_tokens": 3114802.0, "completions/mean_length": 172.5, "completions/min_length": 169.0, "completions/max_length": 181.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 172.5, "completions/min_terminated_length": 169.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.8824431896209717, "rewards/meter/std": 0.08758484572172165, "rewards/count_adherence/mean": 0.9642857313156128, "rewards/count_adherence/std": 0.06613000482320786, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8559232950210571, "rewards/total_composite/std": 0.13603122532367706, "reward": 0.8559232950210571, "reward_std": 0.13603121042251587, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007419355679303408, "sampling/sampling_logp_difference/max": 2.97648024559021, "sampling/importance_sampling_ratio/min": 0.050971925258636475, "sampling/importance_sampling_ratio/mean": 0.9980747103691101, "sampling/importance_sampling_ratio/max": 1.4254961013793945, "entropy": 0.012031709309667349, "clip_ratio/low_mean": 0.0007062146905809641, "clip_ratio/low_min": 0.0007062146905809641, "clip_ratio/high_mean": 0.006558730383403599, "clip_ratio/high_max": 0.006558730383403599, "clip_ratio/region_mean": 0.007264945073984563, "reward_total_mean": 0.8559232950210571, "reward_meter_mean": 0.8824431896209717, "reward_meter_std": 0.08758484572172165, "reward_count_adherence_mean": 0.9642857313156128, "reward_count_adherence_std": 0.06613000482320786, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8559232950210571, "reward_total_composite_std": 0.13603122532367706, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1433.0} {"timestamp_utc": "2026-04-11T22:27:12Z", "mode": "train", "global_step": 1434, "epoch": 0.05537534754402224, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.657575757575759e-06, "num_tokens": 3116978.0, "completions/mean_length": 117.0, "completions/min_length": 117.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.0, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9937511086463928, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6625007390975952, "rewards/total_composite/std": 0.0, "reward": 0.6625007390975952, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 3.828485932899639e-05, "sampling/sampling_logp_difference/max": 0.0012642510700970888, "sampling/importance_sampling_ratio/min": 0.9998624324798584, "sampling/importance_sampling_ratio/mean": 1.0000373125076294, "sampling/importance_sampling_ratio/max": 1.001265048980713, "entropy": 0.000359461108018877, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.6625007390975952, "reward_meter_mean": 0.9937511086463928, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6625007390975952, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1434.0} {"timestamp_utc": "2026-04-11T22:27:17Z", "mode": "train", "global_step": 1435, "epoch": 0.055413963546493666, "loss": -0.0144, "grad_norm": 1.8456981182098389, "learning_rate": 5.654545454545455e-06, "num_tokens": 3119128.0, "completions/mean_length": 86.75, "completions/min_length": 85.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.75, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9707188606262207, "rewards/meter/std": 0.00443111639469862, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9707188606262207, "rewards/total_composite/std": 0.00443111639469862, "reward": 0.9707188606262207, "reward_std": 0.004431125242263079, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008142327889800072, "sampling/sampling_logp_difference/max": 0.5946842432022095, "sampling/importance_sampling_ratio/min": 0.5517367720603943, "sampling/importance_sampling_ratio/mean": 0.9995978474617004, "sampling/importance_sampling_ratio/max": 1.3233660459518433, "entropy": 0.03804446244612336, "clip_ratio/low_mean": 0.011561473016627133, "clip_ratio/low_min": 0.011561473016627133, "clip_ratio/high_mean": 0.002747252816334367, "clip_ratio/high_max": 0.002747252816334367, "clip_ratio/region_mean": 0.0143087258329615, "reward_total_mean": 0.9707188606262207, "reward_meter_mean": 0.9707188606262207, "reward_meter_std": 0.00443111639469862, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9707188606262207, "reward_total_composite_std": 0.00443111639469862, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1435.0} {"timestamp_utc": "2026-04-11T22:27:22Z", "mode": "train", "global_step": 1436, "epoch": 0.05545257954896509, "loss": 0.001, "grad_norm": 0.36797982454299927, "learning_rate": 5.651515151515152e-06, "num_tokens": 3120839.0, "completions/mean_length": 60.875, "completions/min_length": 60.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9974446892738342, "rewards/meter/std": 1.180111758003477e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9974446892738342, "rewards/total_composite/std": 1.180111758003477e-05, "reward": 0.9974446892738342, "reward_std": 1.180111758003477e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008409547619521618, "sampling/sampling_logp_difference/max": 1.4688351154327393, "sampling/importance_sampling_ratio/min": 0.23019348084926605, "sampling/importance_sampling_ratio/mean": 0.9974175691604614, "sampling/importance_sampling_ratio/max": 1.168818473815918, "entropy": 0.017326763598248363, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/high_mean": 0.0020833334419876337, "clip_ratio/high_max": 0.0020833334419876337, "clip_ratio/region_mean": 0.006181693868711591, "reward_total_mean": 0.9974446892738342, "reward_meter_mean": 0.9974446892738342, "reward_meter_std": 1.180111758003477e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9974446892738342, "reward_total_composite_std": 1.180111758003477e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1436.0} {"timestamp_utc": "2026-04-11T22:27:32Z", "mode": "train", "global_step": 1437, "epoch": 0.055491195551436515, "loss": 0.2369, "grad_norm": 1.6375844478607178, "learning_rate": 5.648484848484849e-06, "num_tokens": 3126262.0, "completions/mean_length": 488.875, "completions/min_length": 482.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 485.5714416503906, "completions/min_terminated_length": 482.0, "completions/max_terminated_length": 496.0, "rewards/meter/mean": 0.999032735824585, "rewards/meter/std": 0.00036250357516109943, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.033407654613256454, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.936586856842041, "rewards/total_composite/std": 0.03317427262663841, "reward": 0.936586856842041, "reward_std": 0.03317428007721901, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006067635025829077, "sampling/sampling_logp_difference/max": 4.739413738250732, "sampling/importance_sampling_ratio/min": 0.008743771351873875, "sampling/importance_sampling_ratio/mean": 1.0012120008468628, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.016376970917917788, "clip_ratio/low_mean": 0.003342687035910785, "clip_ratio/low_min": 0.003342687035910785, "clip_ratio/high_mean": 0.0010357403079979122, "clip_ratio/high_max": 0.0010357403079979122, "clip_ratio/region_mean": 0.004378427343908697, "reward_total_mean": 0.936586856842041, "reward_meter_mean": 0.999032735824585, "reward_meter_std": 0.00036250357516109943, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.033407654613256454, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.936586856842041, "reward_total_composite_std": 0.03317427262663841, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1437.0} {"timestamp_utc": "2026-04-11T22:27:38Z", "mode": "train", "global_step": 1438, "epoch": 0.05552981155390794, "loss": 0.0077, "grad_norm": 1.6421748399734497, "learning_rate": 5.645454545454546e-06, "num_tokens": 3129318.0, "completions/mean_length": 182.0, "completions/min_length": 181.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 182.0, "completions/min_terminated_length": 181.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.9796705842018127, "rewards/meter/std": 0.0019595574121922255, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.734752893447876, "rewards/total_composite/std": 0.0014696500729769468, "reward": 0.734752893447876, "reward_std": 0.0014696541475132108, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002134338952600956, "sampling/sampling_logp_difference/max": 0.3413069248199463, "sampling/importance_sampling_ratio/min": 0.7108407020568848, "sampling/importance_sampling_ratio/mean": 1.0006507635116577, "sampling/importance_sampling_ratio/max": 1.3564786911010742, "entropy": 0.011790149379521608, "clip_ratio/low_mean": 0.0006756756920367479, "clip_ratio/low_min": 0.0006756756920367479, "clip_ratio/high_mean": 0.0006868132040835917, "clip_ratio/high_max": 0.0006868132040835917, "clip_ratio/region_mean": 0.0013624888961203396, "reward_total_mean": 0.734752893447876, "reward_meter_mean": 0.9796705842018127, "reward_meter_std": 0.0019595574121922255, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.734752893447876, "reward_total_composite_std": 0.0014696500729769468, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1438.0} {"timestamp_utc": "2026-04-11T22:27:43Z", "mode": "train", "global_step": 1439, "epoch": 0.05556842755637936, "loss": 0.0055, "grad_norm": 0.973854124546051, "learning_rate": 5.642424242424242e-06, "num_tokens": 3131552.0, "completions/mean_length": 91.25, "completions/min_length": 91.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.25, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9798471927642822, "rewards/meter/std": 0.0023329544346779585, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9798471927642822, "rewards/total_composite/std": 0.0023329544346779585, "reward": 0.9798471927642822, "reward_std": 0.0023329604882746935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002926706336438656, "sampling/sampling_logp_difference/max": 0.9269109964370728, "sampling/importance_sampling_ratio/min": 0.39577439427375793, "sampling/importance_sampling_ratio/mean": 1.0003459453582764, "sampling/importance_sampling_ratio/max": 1.3549777269363403, "entropy": 0.011318770702928305, "clip_ratio/low_mean": 0.0013440860202535987, "clip_ratio/low_min": 0.0013440860202535987, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0013440860202535987, "reward_total_mean": 0.9798471927642822, "reward_meter_mean": 0.9798471927642822, "reward_meter_std": 0.0023329544346779585, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9798471927642822, "reward_total_composite_std": 0.0023329544346779585, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1439.0} {"timestamp_utc": "2026-04-11T22:27:50Z", "mode": "train", "global_step": 1440, "epoch": 0.05560704355885079, "loss": 0.0149, "grad_norm": 2.1473212242126465, "learning_rate": 5.6393939393939405e-06, "num_tokens": 3134403.0, "completions/mean_length": 170.375, "completions/min_length": 160.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.375, "completions/min_terminated_length": 160.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.874199390411377, "rewards/meter/std": 0.2657926380634308, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8249521255493164, "rewards/total_composite/std": 0.2578634023666382, "reward": 0.8249521255493164, "reward_std": 0.2578634023666382, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02733701467514038, "sampling/sampling_logp_difference/max": 2.338229179382324, "sampling/importance_sampling_ratio/min": 0.09649837017059326, "sampling/importance_sampling_ratio/mean": 1.001365303993225, "sampling/importance_sampling_ratio/max": 1.8628908395767212, "entropy": 0.1714823842048645, "clip_ratio/low_mean": 0.00674189836718142, "clip_ratio/low_min": 0.00674189836718142, "clip_ratio/high_mean": 0.015878427075222135, "clip_ratio/high_max": 0.015878427075222135, "clip_ratio/region_mean": 0.022620325442403555, "reward_total_mean": 0.8249521255493164, "reward_meter_mean": 0.874199390411377, "reward_meter_std": 0.2657926380634308, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8249521255493164, "reward_total_composite_std": 0.2578634023666382, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1440.0} {"timestamp_utc": "2026-04-11T22:28:00Z", "mode": "train", "global_step": 1441, "epoch": 0.05564565956132221, "loss": -0.0069, "grad_norm": 0.9217522740364075, "learning_rate": 5.636363636363636e-06, "num_tokens": 3139886.0, "completions/mean_length": 473.375, "completions/min_length": 450.0, "completions/max_length": 481.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 473.375, "completions/min_terminated_length": 450.0, "completions/max_terminated_length": 481.0, "rewards/meter/mean": 0.9992083311080933, "rewards/meter/std": 0.0005693559651263058, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9992083311080933, "rewards/total_composite/std": 0.0005693559651263058, "reward": 0.9992083311080933, "reward_std": 0.0005693580606020987, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004486306570470333, "sampling/sampling_logp_difference/max": 4.1551337242126465, "sampling/importance_sampling_ratio/min": 0.015683693811297417, "sampling/importance_sampling_ratio/mean": 0.9997328519821167, "sampling/importance_sampling_ratio/max": 1.7880674600601196, "entropy": 0.01629194081760943, "clip_ratio/low_mean": 0.0008237959118559957, "clip_ratio/low_min": 0.0008237959118559957, "clip_ratio/high_mean": 0.002349771384615451, "clip_ratio/high_max": 0.002349771384615451, "clip_ratio/region_mean": 0.0031735672964714468, "reward_total_mean": 0.9992083311080933, "reward_meter_mean": 0.9992083311080933, "reward_meter_std": 0.0005693559651263058, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9992083311080933, "reward_total_composite_std": 0.0005693559651263058, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1441.0} {"timestamp_utc": "2026-04-11T22:28:05Z", "mode": "train", "global_step": 1442, "epoch": 0.055684275563793635, "loss": -0.0431, "grad_norm": 6.049340724945068, "learning_rate": 5.633333333333334e-06, "num_tokens": 3141822.0, "completions/mean_length": 68.0, "completions/min_length": 57.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9196300506591797, "rewards/meter/std": 0.19259606301784515, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9196300506591797, "rewards/total_composite/std": 0.19259606301784515, "reward": 0.9196300506591797, "reward_std": 0.19259603321552277, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03470028191804886, "sampling/sampling_logp_difference/max": 2.3933558464050293, "sampling/importance_sampling_ratio/min": 0.09132270514965057, "sampling/importance_sampling_ratio/mean": 0.9963318109512329, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08851629169657826, "clip_ratio/low_mean": 0.0021929824724793434, "clip_ratio/low_min": 0.0021929824724793434, "clip_ratio/high_mean": 0.036970873130485415, "clip_ratio/high_max": 0.036970873130485415, "clip_ratio/region_mean": 0.03916385560296476, "reward_total_mean": 0.9196300506591797, "reward_meter_mean": 0.9196300506591797, "reward_meter_std": 0.19259606301784515, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9196300506591797, "reward_total_composite_std": 0.19259606301784515, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1442.0} {"timestamp_utc": "2026-04-11T22:28:10Z", "mode": "train", "global_step": 1443, "epoch": 0.05572289156626506, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.630303030303031e-06, "num_tokens": 3144326.0, "completions/mean_length": 130.0, "completions/min_length": 130.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.0, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9937425255775452, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7453069090843201, "rewards/total_composite/std": 0.0, "reward": 0.7453069090843201, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005708218086510897, "sampling/sampling_logp_difference/max": 0.12230955064296722, "sampling/importance_sampling_ratio/min": 0.9506359696388245, "sampling/importance_sampling_ratio/mean": 1.0004909038543701, "sampling/importance_sampling_ratio/max": 1.1301038265228271, "entropy": 0.006344929046463221, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7453069090843201, "reward_meter_mean": 0.9937425255775452, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7453069090843201, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1443.0} {"timestamp_utc": "2026-04-11T22:28:15Z", "mode": "train", "global_step": 1444, "epoch": 0.05576150756873648, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.627272727272728e-06, "num_tokens": 3145846.0, "completions/mean_length": 31.0, "completions/min_length": 31.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9980738162994385, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9980738162994385, "rewards/total_composite/std": 0.0, "reward": 0.9980738162994385, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.002949180081486702, "sampling/sampling_logp_difference/max": 0.04289769381284714, "sampling/importance_sampling_ratio/min": 0.9580094218254089, "sampling/importance_sampling_ratio/mean": 1.0023293495178223, "sampling/importance_sampling_ratio/max": 1.035339593887329, "entropy": 0.026092080865055323, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9980738162994385, "reward_meter_mean": 0.9980738162994385, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9980738162994385, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1444.0} {"timestamp_utc": "2026-04-11T22:28:22Z", "mode": "train", "global_step": 1445, "epoch": 0.05580012357120791, "loss": 0.06, "grad_norm": 2.486631155014038, "learning_rate": 5.624242424242424e-06, "num_tokens": 3149433.0, "completions/mean_length": 243.375, "completions/min_length": 225.0, "completions/max_length": 297.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 243.375, "completions/min_terminated_length": 225.0, "completions/max_terminated_length": 297.0, "rewards/meter/mean": 0.8477659821510315, "rewards/meter/std": 0.20163950324058533, "rewards/count_adherence/mean": 0.953125, "rewards/count_adherence/std": 0.06469365209341049, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8106421232223511, "rewards/total_composite/std": 0.20955340564250946, "reward": 0.8106421232223511, "reward_std": 0.20955340564250946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014041267335414886, "sampling/sampling_logp_difference/max": 1.3953499794006348, "sampling/importance_sampling_ratio/min": 0.2477463185787201, "sampling/importance_sampling_ratio/mean": 1.002934455871582, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07382735377177596, "clip_ratio/low_mean": 0.003169349074596539, "clip_ratio/low_min": 0.003169349074596539, "clip_ratio/high_mean": 0.009951379150152206, "clip_ratio/high_max": 0.009951379150152206, "clip_ratio/region_mean": 0.013120728224748746, "reward_total_mean": 0.8106421232223511, "reward_meter_mean": 0.8477659821510315, "reward_meter_std": 0.20163950324058533, "reward_count_adherence_mean": 0.953125, "reward_count_adherence_std": 0.06469365209341049, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8106421232223511, "reward_total_composite_std": 0.20955340564250946, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1445.0} {"timestamp_utc": "2026-04-11T22:28:28Z", "mode": "train", "global_step": 1446, "epoch": 0.05583873957367933, "loss": 0.0042, "grad_norm": 1.7898367643356323, "learning_rate": 5.6212121212121215e-06, "num_tokens": 3152287.0, "completions/mean_length": 175.75, "completions/min_length": 161.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 175.75, "completions/min_terminated_length": 161.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9380402565002441, "rewards/meter/std": 0.1394338458776474, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7035301923751831, "rewards/total_composite/std": 0.10457538068294525, "reward": 0.7035301923751831, "reward_std": 0.10457538813352585, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016932779923081398, "sampling/sampling_logp_difference/max": 1.4186067581176758, "sampling/importance_sampling_ratio/min": 0.24205103516578674, "sampling/importance_sampling_ratio/mean": 1.0033235549926758, "sampling/importance_sampling_ratio/max": 1.9580739736557007, "entropy": 0.07765469187870622, "clip_ratio/low_mean": 0.0014204545877873898, "clip_ratio/low_min": 0.0014204545877873898, "clip_ratio/high_mean": 0.008528741833288223, "clip_ratio/high_max": 0.008528741833288223, "clip_ratio/region_mean": 0.009949196421075612, "reward_total_mean": 0.7035301923751831, "reward_meter_mean": 0.9380402565002441, "reward_meter_std": 0.1394338458776474, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7035301923751831, "reward_total_composite_std": 0.10457538068294525, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1446.0} {"timestamp_utc": "2026-04-11T22:28:33Z", "mode": "train", "global_step": 1447, "epoch": 0.055877355576150756, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.618181818181818e-06, "num_tokens": 3154039.0, "completions/mean_length": 65.0, "completions/min_length": 65.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9946109652519226, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9946109652519226, "rewards/total_composite/std": 0.0, "reward": 0.9946109652519226, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0004363986663520336, "sampling/sampling_logp_difference/max": 0.07490222156047821, "sampling/importance_sampling_ratio/min": 0.9278342127799988, "sampling/importance_sampling_ratio/mean": 0.9998361468315125, "sampling/importance_sampling_ratio/max": 1.0040090084075928, "entropy": 0.0026015231778728776, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9946109652519226, "reward_meter_mean": 0.9946109652519226, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9946109652519226, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1447.0} {"timestamp_utc": "2026-04-11T22:28:38Z", "mode": "train", "global_step": 1448, "epoch": 0.05591597157862218, "loss": 0.0755, "grad_norm": 8.518555641174316, "learning_rate": 5.615151515151516e-06, "num_tokens": 3155586.0, "completions/mean_length": 35.375, "completions/min_length": 31.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9522101283073425, "rewards/meter/std": 0.04017549753189087, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9522101283073425, "rewards/total_composite/std": 0.04017549753189087, "reward": 0.9522101283073425, "reward_std": 0.04017549380660057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05455894395709038, "sampling/sampling_logp_difference/max": 2.4194202423095703, "sampling/importance_sampling_ratio/min": 0.08897318691015244, "sampling/importance_sampling_ratio/mean": 0.9952448606491089, "sampling/importance_sampling_ratio/max": 1.8271827697753906, "entropy": 0.2404468934983015, "clip_ratio/low_mean": 0.013591269962489605, "clip_ratio/low_min": 0.013591269962489605, "clip_ratio/high_mean": 0.01865279395133257, "clip_ratio/high_max": 0.01865279395133257, "clip_ratio/region_mean": 0.032244063913822174, "reward_total_mean": 0.9522101283073425, "reward_meter_mean": 0.9522101283073425, "reward_meter_std": 0.04017549753189087, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9522101283073425, "reward_total_composite_std": 0.04017549753189087, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1448.0} {"timestamp_utc": "2026-04-11T22:28:42Z", "mode": "train", "global_step": 1449, "epoch": 0.055954587581093604, "loss": 0.0373, "grad_norm": 10.865769386291504, "learning_rate": 5.612121212121212e-06, "num_tokens": 3156985.0, "completions/mean_length": 31.875, "completions/min_length": 30.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.5450050234794617, "rewards/meter/std": 0.40841594338417053, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.5450050234794617, "rewards/total_composite/std": 0.40841594338417053, "reward": 0.5450050234794617, "reward_std": 0.40841594338417053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.050434429198503494, "sampling/sampling_logp_difference/max": 2.27243709564209, "sampling/importance_sampling_ratio/min": 0.10306069999933243, "sampling/importance_sampling_ratio/mean": 0.9940862059593201, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10359709989279509, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.016137432772666216, "clip_ratio/high_max": 0.016137432772666216, "clip_ratio/region_mean": 0.02371319057419896, "reward_total_mean": 0.5450050234794617, "reward_meter_mean": 0.5450050234794617, "reward_meter_std": 0.40841594338417053, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.5450050234794617, "reward_total_composite_std": 0.40841594338417053, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1449.0} {"timestamp_utc": "2026-04-11T22:28:49Z", "mode": "train", "global_step": 1450, "epoch": 0.05599320358356503, "loss": -0.0266, "grad_norm": 1.0666146278381348, "learning_rate": 5.60909090909091e-06, "num_tokens": 3160875.0, "completions/mean_length": 289.25, "completions/min_length": 273.0, "completions/max_length": 309.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 289.25, "completions/min_terminated_length": 273.0, "completions/max_terminated_length": 309.0, "rewards/meter/mean": 0.9899704456329346, "rewards/meter/std": 0.010459767654538155, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9899704456329346, "rewards/total_composite/std": 0.010459767654538155, "reward": 0.9899704456329346, "reward_std": 0.010459772311151028, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017150694504380226, "sampling/sampling_logp_difference/max": 3.746622323989868, "sampling/importance_sampling_ratio/min": 0.023597314953804016, "sampling/importance_sampling_ratio/mean": 0.9994972944259644, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05681949155405164, "clip_ratio/low_mean": 0.00451769056962803, "clip_ratio/low_min": 0.00451769056962803, "clip_ratio/high_mean": 0.011592184397159144, "clip_ratio/high_max": 0.011592184397159144, "clip_ratio/region_mean": 0.016109874966787174, "reward_total_mean": 0.9899704456329346, "reward_meter_mean": 0.9899704456329346, "reward_meter_std": 0.010459767654538155, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9899704456329346, "reward_total_composite_std": 0.010459767654538155, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1450.0} {"timestamp_utc": "2026-04-11T22:30:17Z", "mode": "eval", "global_step": 1450, "epoch": 0.05599320358356503, "eval_loss": NaN, "eval_runtime": 87.9807, "eval_samples_per_second": 1.182, "eval_steps_per_second": 0.148, "eval_num_tokens": 3160875.0, "eval_completions/mean_length": 239.92307692307693, "eval_completions/min_length": 66.15384615384616, "eval_completions/max_length": 467.46153846153845, "eval_completions/clipped_ratio": 0.10576923076923077, "eval_completions/mean_terminated_length": 206.28709294245795, "eval_completions/min_terminated_length": 66.15384615384616, "eval_completions/max_terminated_length": 399.61538461538464, "eval_rewards/meter/mean": 0.6795406754200275, "eval_rewards/meter/std": 0.3748341420522103, "eval_rewards/count_adherence/mean": 0.8663222560515771, "eval_rewards/count_adherence/std": 0.14821005669923928, "eval_rewards/arabic_clean/mean": 0.9615384615384616, "eval_rewards/arabic_clean/std": 0.10878565678229699, "eval_rewards/total_composite/mean": 0.5920751897188333, "eval_rewards/total_composite/std": 0.36964805538837725, "eval_reward": 0.5920751897188333, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.00830321037210524, "eval_sampling/sampling_logp_difference/max": 0.984417227598337, "eval_sampling/importance_sampling_ratio/min": 0.40910858145126927, "eval_sampling/importance_sampling_ratio/mean": 1.0022475260954637, "eval_sampling/importance_sampling_ratio/max": 1.411508138363178, "eval_entropy": 0.06675205623301175, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5920751897188333, "eval_reward_meter_mean": 0.6795406754200275, "eval_reward_meter_std": 0.3748341420522103, "eval_reward_count_adherence_mean": 0.8663222560515771, "eval_reward_count_adherence_std": 0.14821005669923928, "eval_reward_arabic_clean_mean": 0.9615384615384616, "eval_reward_arabic_clean_std": 0.10878565678229699, "eval_reward_total_composite_mean": 0.5920751897188333, "eval_reward_total_composite_std": 0.36964805538837725, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1450.0} {"timestamp_utc": "2026-04-11T22:30:25Z", "mode": "train", "global_step": 1451, "epoch": 0.05603181958603645, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.606060606060606e-06, "num_tokens": 3162907.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9806720614433289, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9806720614433289, "rewards/total_composite/std": 0.0, "reward": 0.9806720614433289, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.001633047591894865, "sampling/sampling_logp_difference/max": 0.356443852186203, "sampling/importance_sampling_ratio/min": 0.7001618146896362, "sampling/importance_sampling_ratio/mean": 0.9994708299636841, "sampling/importance_sampling_ratio/max": 1.0457462072372437, "entropy": 0.007809586415532976, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9806720614433289, "reward_meter_mean": 0.9806720614433289, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9806720614433289, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1451.0} {"timestamp_utc": "2026-04-11T22:30:31Z", "mode": "train", "global_step": 1452, "epoch": 0.056070435588507876, "loss": -0.0247, "grad_norm": 2.475581407546997, "learning_rate": 5.603030303030303e-06, "num_tokens": 3165491.0, "completions/mean_length": 136.0, "completions/min_length": 131.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.0, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9831618666648865, "rewards/meter/std": 0.0138022406026721, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6554412841796875, "rewards/total_composite/std": 0.00920148566365242, "reward": 0.6554412841796875, "reward_std": 0.009201486594974995, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019362740218639374, "sampling/sampling_logp_difference/max": 1.7038183212280273, "sampling/importance_sampling_ratio/min": 0.18198730051517487, "sampling/importance_sampling_ratio/mean": 1.000468134880066, "sampling/importance_sampling_ratio/max": 1.8243868350982666, "entropy": 0.07754070544615388, "clip_ratio/low_mean": 0.0028625954291783273, "clip_ratio/low_min": 0.0028625954291783273, "clip_ratio/high_mean": 0.012375308782793581, "clip_ratio/high_max": 0.012375308782793581, "clip_ratio/region_mean": 0.015237904211971909, "reward_total_mean": 0.6554412841796875, "reward_meter_mean": 0.9831618666648865, "reward_meter_std": 0.0138022406026721, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6554412841796875, "reward_total_composite_std": 0.00920148566365242, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1452.0} {"timestamp_utc": "2026-04-11T22:30:36Z", "mode": "train", "global_step": 1453, "epoch": 0.0561090515909793, "loss": 0.0129, "grad_norm": 4.42448616027832, "learning_rate": 5.600000000000001e-06, "num_tokens": 3167506.0, "completions/mean_length": 72.875, "completions/min_length": 70.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.875, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9939925670623779, "rewards/meter/std": 0.0024767920840531588, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9939925670623779, "rewards/total_composite/std": 0.0024767920840531588, "reward": 0.9939925670623779, "reward_std": 0.0024768023286014795, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03453652933239937, "sampling/sampling_logp_difference/max": 1.8276549577713013, "sampling/importance_sampling_ratio/min": 0.16079019010066986, "sampling/importance_sampling_ratio/mean": 1.0055344104766846, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10995942819863558, "clip_ratio/low_mean": 0.020516035030595958, "clip_ratio/low_min": 0.020516035030595958, "clip_ratio/high_mean": 0.015577435493469238, "clip_ratio/high_max": 0.015577435493469238, "clip_ratio/region_mean": 0.036093470524065197, "reward_total_mean": 0.9939925670623779, "reward_meter_mean": 0.9939925670623779, "reward_meter_std": 0.0024767920840531588, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9939925670623779, "reward_total_composite_std": 0.0024767920840531588, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1453.0} {"timestamp_utc": "2026-04-11T22:30:43Z", "mode": "train", "global_step": 1454, "epoch": 0.056147667593450724, "loss": -0.0103, "grad_norm": 4.566795349121094, "learning_rate": 5.596969696969697e-06, "num_tokens": 3170916.0, "completions/mean_length": 237.25, "completions/min_length": 226.0, "completions/max_length": 254.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 237.25, "completions/min_terminated_length": 226.0, "completions/max_terminated_length": 254.0, "rewards/meter/mean": 0.9813140630722046, "rewards/meter/std": 0.012663300149142742, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8177617788314819, "rewards/total_composite/std": 0.010552753694355488, "reward": 0.8177617788314819, "reward_std": 0.01055274810642004, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011705584824085236, "sampling/sampling_logp_difference/max": 1.739227533340454, "sampling/importance_sampling_ratio/min": 0.1756560355424881, "sampling/importance_sampling_ratio/mean": 0.9996017813682556, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03703577653504908, "clip_ratio/low_mean": 0.0032744554919190705, "clip_ratio/low_min": 0.0032744554919190705, "clip_ratio/high_mean": 0.0047182978596538305, "clip_ratio/high_max": 0.0047182978596538305, "clip_ratio/region_mean": 0.007992753351572901, "reward_total_mean": 0.8177617788314819, "reward_meter_mean": 0.9813140630722046, "reward_meter_std": 0.012663300149142742, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8177617788314819, "reward_total_composite_std": 0.010552753694355488, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1454.0} {"timestamp_utc": "2026-04-11T22:30:49Z", "mode": "train", "global_step": 1455, "epoch": 0.05618628359592215, "loss": -0.0013, "grad_norm": 1.4705860614776611, "learning_rate": 5.593939393939395e-06, "num_tokens": 3174132.0, "completions/mean_length": 213.0, "completions/min_length": 212.0, "completions/max_length": 214.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 213.0, "completions/min_terminated_length": 212.0, "completions/max_terminated_length": 214.0, "rewards/meter/mean": 0.9967323541641235, "rewards/meter/std": 0.0007932247826829553, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9967323541641235, "rewards/total_composite/std": 0.0007932247826829553, "reward": 0.9967323541641235, "reward_std": 0.0007932398002594709, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004804201424121857, "sampling/sampling_logp_difference/max": 1.4114460945129395, "sampling/importance_sampling_ratio/min": 0.24379049241542816, "sampling/importance_sampling_ratio/mean": 0.999669075012207, "sampling/importance_sampling_ratio/max": 1.6204928159713745, "entropy": 0.015388465602882206, "clip_ratio/low_mean": 0.000589622650295496, "clip_ratio/low_min": 0.000589622650295496, "clip_ratio/high_mean": 0.004689351073466241, "clip_ratio/high_max": 0.004689351073466241, "clip_ratio/region_mean": 0.005278973723761737, "reward_total_mean": 0.9967323541641235, "reward_meter_mean": 0.9967323541641235, "reward_meter_std": 0.0007932247826829553, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9967323541641235, "reward_total_composite_std": 0.0007932247826829553, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1455.0} {"timestamp_utc": "2026-04-11T22:30:56Z", "mode": "train", "global_step": 1456, "epoch": 0.05622489959839357, "loss": 0.0045, "grad_norm": 2.265878915786743, "learning_rate": 5.5909090909090915e-06, "num_tokens": 3178004.0, "completions/mean_length": 235.0, "completions/min_length": 218.0, "completions/max_length": 256.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 235.0, "completions/min_terminated_length": 218.0, "completions/max_terminated_length": 256.0, "rewards/meter/mean": 0.8829516172409058, "rewards/meter/std": 0.12476588785648346, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.06681530922651291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8259998559951782, "rewards/total_composite/std": 0.11845896393060684, "reward": 0.8259998559951782, "reward_std": 0.11845897138118744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03738979622721672, "sampling/sampling_logp_difference/max": 2.805210828781128, "sampling/importance_sampling_ratio/min": 0.060494013130664825, "sampling/importance_sampling_ratio/mean": 1.0086950063705444, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26418917160481215, "clip_ratio/low_mean": 0.007535498822107911, "clip_ratio/low_min": 0.007535498822107911, "clip_ratio/high_mean": 0.02057173242792487, "clip_ratio/high_max": 0.02057173242792487, "clip_ratio/region_mean": 0.028107231250032783, "reward_total_mean": 0.8259998559951782, "reward_meter_mean": 0.8829516172409058, "reward_meter_std": 0.12476588785648346, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.06681530922651291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8259998559951782, "reward_total_composite_std": 0.11845896393060684, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1456.0} {"timestamp_utc": "2026-04-11T22:31:01Z", "mode": "train", "global_step": 1457, "epoch": 0.056263515600865, "loss": 0.0211, "grad_norm": 7.5675835609436035, "learning_rate": 5.587878787878789e-06, "num_tokens": 3179941.0, "completions/mean_length": 74.125, "completions/min_length": 70.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.125, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.997861385345459, "rewards/meter/std": 0.0013210211182013154, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.997861385345459, "rewards/total_composite/std": 0.0013210211182013154, "reward": 0.997861385345459, "reward_std": 0.0013210255419835448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07839632779359818, "sampling/sampling_logp_difference/max": 7.479003429412842, "sampling/importance_sampling_ratio/min": 0.000564820074941963, "sampling/importance_sampling_ratio/mean": 0.9866158366203308, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11953066289424896, "clip_ratio/low_mean": 0.0016666667070239782, "clip_ratio/low_min": 0.0016666667070239782, "clip_ratio/high_mean": 0.025570699479430914, "clip_ratio/high_max": 0.025570699479430914, "clip_ratio/region_mean": 0.027237366186454892, "reward_total_mean": 0.997861385345459, "reward_meter_mean": 0.997861385345459, "reward_meter_std": 0.0013210211182013154, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.997861385345459, "reward_total_composite_std": 0.0013210211182013154, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1457.0} {"timestamp_utc": "2026-04-11T22:31:11Z", "mode": "train", "global_step": 1458, "epoch": 0.05630213160333642, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.584848484848485e-06, "num_tokens": 3181581.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.5341699123382568, "rewards/meter/std": 0.40559667348861694, "rewards/count_adherence/mean": 0.7946428656578064, "rewards/count_adherence/std": 0.08903024345636368, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.45501649379730225, "rewards/total_composite/std": 0.36901408433914185, "reward": 0.45501649379730225, "reward_std": 0.36901408433914185, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.45501649379730225, "reward_meter_mean": 0.5341699123382568, "reward_meter_std": 0.40559667348861694, "reward_count_adherence_mean": 0.7946428656578064, "reward_count_adherence_std": 0.08903024345636368, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.45501649379730225, "reward_total_composite_std": 0.36901408433914185, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1458.0} {"timestamp_utc": "2026-04-11T22:31:16Z", "mode": "train", "global_step": 1459, "epoch": 0.056340747605807845, "loss": -0.1253, "grad_norm": 4.969435691833496, "learning_rate": 5.5818181818181824e-06, "num_tokens": 3183560.0, "completions/mean_length": 83.375, "completions/min_length": 62.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.375, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.8118997812271118, "rewards/meter/std": 0.34411779046058655, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8118997812271118, "rewards/total_composite/std": 0.34411779046058655, "reward": 0.8118997812271118, "reward_std": 0.34411782026290894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042707011103630066, "sampling/sampling_logp_difference/max": 4.485851287841797, "sampling/importance_sampling_ratio/min": 0.011267292313277721, "sampling/importance_sampling_ratio/mean": 1.0002074241638184, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14178061299026012, "clip_ratio/low_mean": 0.011968766339123249, "clip_ratio/low_min": 0.011968766339123249, "clip_ratio/high_mean": 0.027695991564542055, "clip_ratio/high_max": 0.027695991564542055, "clip_ratio/region_mean": 0.039664757903665304, "reward_total_mean": 0.8118997812271118, "reward_meter_mean": 0.8118997812271118, "reward_meter_std": 0.34411779046058655, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8118997812271118, "reward_total_composite_std": 0.34411779046058655, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1459.0} {"timestamp_utc": "2026-04-11T22:31:21Z", "mode": "train", "global_step": 1460, "epoch": 0.05637936360827927, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.578787878787879e-06, "num_tokens": 3186032.0, "completions/mean_length": 145.0, "completions/min_length": 145.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.0, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9806720614433289, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6537813544273376, "rewards/total_composite/std": 0.0, "reward": 0.6537813544273376, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006778101669624448, "sampling/sampling_logp_difference/max": 0.02597404271364212, "sampling/importance_sampling_ratio/min": 0.9746548533439636, "sampling/importance_sampling_ratio/mean": 1.000344157218933, "sampling/importance_sampling_ratio/max": 1.026314377784729, "entropy": 0.006548340024892241, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.6537813544273376, "reward_meter_mean": 0.9806720614433289, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6537813544273376, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1460.0} {"timestamp_utc": "2026-04-11T22:31:26Z", "mode": "train", "global_step": 1461, "epoch": 0.05641797961075069, "loss": -0.0227, "grad_norm": 11.338554382324219, "learning_rate": 5.575757575757577e-06, "num_tokens": 3187607.0, "completions/mean_length": 34.875, "completions/min_length": 33.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7105430364608765, "rewards/meter/std": 0.4016382396221161, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7105430364608765, "rewards/total_composite/std": 0.4016382396221161, "reward": 0.7105430364608765, "reward_std": 0.4016382396221161, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04106371849775314, "sampling/sampling_logp_difference/max": 0.967343807220459, "sampling/importance_sampling_ratio/min": 0.3800913095474243, "sampling/importance_sampling_ratio/mean": 1.0075314044952393, "sampling/importance_sampling_ratio/max": 1.5996967554092407, "entropy": 0.22307384572923183, "clip_ratio/low_mean": 0.018506493885070086, "clip_ratio/low_min": 0.018506493885070086, "clip_ratio/high_mean": 0.034936722135171294, "clip_ratio/high_max": 0.034936722135171294, "clip_ratio/region_mean": 0.05344321602024138, "reward_total_mean": 0.7105430364608765, "reward_meter_mean": 0.7105430364608765, "reward_meter_std": 0.4016382396221161, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7105430364608765, "reward_total_composite_std": 0.4016382396221161, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1461.0} {"timestamp_utc": "2026-04-11T22:31:31Z", "mode": "train", "global_step": 1462, "epoch": 0.05645659561322212, "loss": -0.0122, "grad_norm": 0.35829195380210876, "learning_rate": 5.572727272727273e-06, "num_tokens": 3189525.0, "completions/mean_length": 74.75, "completions/min_length": 73.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.75, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9802955389022827, "rewards/meter/std": 0.0012241577496752143, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9802955389022827, "rewards/total_composite/std": 0.0012241577496752143, "reward": 0.9802955389022827, "reward_std": 0.0012241617077961564, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007379383314400911, "sampling/sampling_logp_difference/max": 1.2330729961395264, "sampling/importance_sampling_ratio/min": 0.29139575362205505, "sampling/importance_sampling_ratio/mean": 0.9975477457046509, "sampling/importance_sampling_ratio/max": 1.1428965330123901, "entropy": 0.014824787271209061, "clip_ratio/low_mean": 0.0034246575087308884, "clip_ratio/low_min": 0.0034246575087308884, "clip_ratio/high_mean": 0.003149110358208418, "clip_ratio/high_max": 0.003149110358208418, "clip_ratio/region_mean": 0.006573767866939306, "reward_total_mean": 0.9802955389022827, "reward_meter_mean": 0.9802955389022827, "reward_meter_std": 0.0012241577496752143, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9802955389022827, "reward_total_composite_std": 0.0012241577496752143, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1462.0} {"timestamp_utc": "2026-04-11T22:31:35Z", "mode": "train", "global_step": 1463, "epoch": 0.05649521161569354, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.569696969696971e-06, "num_tokens": 3191133.0, "completions/mean_length": 37.0, "completions/min_length": 37.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9806720614433289, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9806720614433289, "rewards/total_composite/std": 0.0, "reward": 0.9806720614433289, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.001456483849324286, "sampling/sampling_logp_difference/max": 0.06053692847490311, "sampling/importance_sampling_ratio/min": 0.9673674702644348, "sampling/importance_sampling_ratio/mean": 1.0012251138687134, "sampling/importance_sampling_ratio/max": 1.0624068975448608, "entropy": 0.013269567512907088, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9806720614433289, "reward_meter_mean": 0.9806720614433289, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9806720614433289, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1463.0} {"timestamp_utc": "2026-04-11T22:31:40Z", "mode": "train", "global_step": 1464, "epoch": 0.056533827618164965, "loss": -0.0136, "grad_norm": 5.499317169189453, "learning_rate": 5.566666666666667e-06, "num_tokens": 3192791.0, "completions/mean_length": 65.25, "completions/min_length": 61.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7975334525108337, "rewards/meter/std": 0.21782991290092468, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7975334525108337, "rewards/total_composite/std": 0.21782991290092468, "reward": 0.7975334525108337, "reward_std": 0.21782991290092468, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021826768293976784, "sampling/sampling_logp_difference/max": 0.881016731262207, "sampling/importance_sampling_ratio/min": 0.4143614172935486, "sampling/importance_sampling_ratio/mean": 0.9999092221260071, "sampling/importance_sampling_ratio/max": 1.5156503915786743, "entropy": 0.12650380190461874, "clip_ratio/low_mean": 0.013766789110377431, "clip_ratio/low_min": 0.013766789110377431, "clip_ratio/high_mean": 0.01518264482729137, "clip_ratio/high_max": 0.01518264482729137, "clip_ratio/region_mean": 0.0289494339376688, "reward_total_mean": 0.7975334525108337, "reward_meter_mean": 0.7975334525108337, "reward_meter_std": 0.21782991290092468, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7975334525108337, "reward_total_composite_std": 0.21782991290092468, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1464.0} {"timestamp_utc": "2026-04-11T22:31:44Z", "mode": "train", "global_step": 1465, "epoch": 0.05657244362063639, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.563636363636364e-06, "num_tokens": 3194439.0, "completions/mean_length": 49.0, "completions/min_length": 49.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9290721416473389, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9290721416473389, "rewards/total_composite/std": 0.0, "reward": 0.9290721416473389, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005562398000620306, "sampling/sampling_logp_difference/max": 0.045216623693704605, "sampling/importance_sampling_ratio/min": 0.9557904005050659, "sampling/importance_sampling_ratio/mean": 1.0001686811447144, "sampling/importance_sampling_ratio/max": 1.0263056755065918, "entropy": 0.004303567809984088, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9290721416473389, "reward_meter_mean": 0.9290721416473389, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9290721416473389, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1465.0} {"timestamp_utc": "2026-04-11T22:31:49Z", "mode": "train", "global_step": 1466, "epoch": 0.056611059623107814, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.560606060606061e-06, "num_tokens": 3196311.0, "completions/mean_length": 65.0, "completions/min_length": 65.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9946109652519226, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9946109652519226, "rewards/total_composite/std": 0.0, "reward": 0.9946109652519226, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 8.761865319684148e-05, "sampling/sampling_logp_difference/max": 0.0034110709093511105, "sampling/importance_sampling_ratio/min": 0.9979086518287659, "sampling/importance_sampling_ratio/mean": 1.0000669956207275, "sampling/importance_sampling_ratio/max": 1.003416895866394, "entropy": 0.0011086041558883153, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9946109652519226, "reward_meter_mean": 0.9946109652519226, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9946109652519226, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1466.0} {"timestamp_utc": "2026-04-11T22:31:54Z", "mode": "train", "global_step": 1467, "epoch": 0.05664967562557924, "loss": 0.0039, "grad_norm": 6.296424865722656, "learning_rate": 5.557575757575758e-06, "num_tokens": 3198750.0, "completions/mean_length": 143.875, "completions/min_length": 137.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.875, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9456251859664917, "rewards/meter/std": 0.10555077344179153, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.630416750907898, "rewards/total_composite/std": 0.07036717981100082, "reward": 0.630416750907898, "reward_std": 0.07036718726158142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005361086688935757, "sampling/sampling_logp_difference/max": 1.4393396377563477, "sampling/importance_sampling_ratio/min": 0.2370842695236206, "sampling/importance_sampling_ratio/mean": 1.0005542039871216, "sampling/importance_sampling_ratio/max": 1.834373116493225, "entropy": 0.01866331562632695, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/high_mean": 0.0017241379246115685, "clip_ratio/high_max": 0.0017241379246115685, "clip_ratio/region_mean": 0.0034602490486577153, "reward_total_mean": 0.630416750907898, "reward_meter_mean": 0.9456251859664917, "reward_meter_std": 0.10555077344179153, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.630416750907898, "reward_total_composite_std": 0.07036717981100082, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1467.0} {"timestamp_utc": "2026-04-11T22:31:59Z", "mode": "train", "global_step": 1468, "epoch": 0.05668829162805066, "loss": -0.0044, "grad_norm": 5.829403877258301, "learning_rate": 5.554545454545454e-06, "num_tokens": 3200904.0, "completions/mean_length": 99.25, "completions/min_length": 90.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.25, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.018819713965058327, "rewards/meter/std": 0.020291732624173164, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.012546476908028126, "rewards/total_composite/std": 0.013527821749448776, "reward": 0.012546476908028126, "reward_std": 0.013527821749448776, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03759616240859032, "sampling/sampling_logp_difference/max": 2.500622034072876, "sampling/importance_sampling_ratio/min": 0.08203395456075668, "sampling/importance_sampling_ratio/mean": 0.9977881908416748, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10704822419211268, "clip_ratio/low_mean": 0.022970885620452464, "clip_ratio/low_min": 0.022970885620452464, "clip_ratio/high_mean": 0.010858586290851235, "clip_ratio/high_max": 0.010858586290851235, "clip_ratio/region_mean": 0.0338294719113037, "reward_total_mean": 0.012546476908028126, "reward_meter_mean": 0.018819713965058327, "reward_meter_std": 0.020291732624173164, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.012546476908028126, "reward_total_composite_std": 0.013527821749448776, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1468.0} {"timestamp_utc": "2026-04-11T22:32:05Z", "mode": "train", "global_step": 1469, "epoch": 0.056726907630522086, "loss": -0.0181, "grad_norm": 4.586915493011475, "learning_rate": 5.5515151515151524e-06, "num_tokens": 3203267.0, "completions/mean_length": 128.375, "completions/min_length": 117.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.375, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.7937608957290649, "rewards/meter/std": 0.21430887281894684, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6724545359611511, "rewards/total_composite/std": 0.21081210672855377, "reward": 0.6724545359611511, "reward_std": 0.21081212162971497, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028733236715197563, "sampling/sampling_logp_difference/max": 2.4430007934570312, "sampling/importance_sampling_ratio/min": 0.08689969033002853, "sampling/importance_sampling_ratio/mean": 1.0030384063720703, "sampling/importance_sampling_ratio/max": 1.8921717405319214, "entropy": 0.15858958289027214, "clip_ratio/low_mean": 0.01010369905270636, "clip_ratio/low_min": 0.01010369905270636, "clip_ratio/high_mean": 0.010604630340822041, "clip_ratio/high_max": 0.010604630340822041, "clip_ratio/region_mean": 0.020708329393528402, "reward_total_mean": 0.6724545359611511, "reward_meter_mean": 0.7937608957290649, "reward_meter_std": 0.21430887281894684, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6724545359611511, "reward_total_composite_std": 0.21081210672855377, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1469.0} {"timestamp_utc": "2026-04-11T22:32:10Z", "mode": "train", "global_step": 1470, "epoch": 0.05676552363299351, "loss": -0.0035, "grad_norm": 1.312452793121338, "learning_rate": 5.548484848484849e-06, "num_tokens": 3205718.0, "completions/mean_length": 133.375, "completions/min_length": 127.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.375, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.9971146583557129, "rewards/meter/std": 4.810225800611079e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9971146583557129, "rewards/total_composite/std": 4.810225800611079e-05, "reward": 0.9971146583557129, "reward_std": 4.809494566870853e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00545355724170804, "sampling/sampling_logp_difference/max": 0.8365278244018555, "sampling/importance_sampling_ratio/min": 0.4332121014595032, "sampling/importance_sampling_ratio/mean": 1.001191258430481, "sampling/importance_sampling_ratio/max": 1.750195860862732, "entropy": 0.02232604706659913, "clip_ratio/low_mean": 0.0019685039296746254, "clip_ratio/low_min": 0.0019685039296746254, "clip_ratio/high_mean": 0.0018587617087177932, "clip_ratio/high_max": 0.0018587617087177932, "clip_ratio/region_mean": 0.0038272656383924186, "reward_total_mean": 0.9971146583557129, "reward_meter_mean": 0.9971146583557129, "reward_meter_std": 4.810225800611079e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9971146583557129, "reward_total_composite_std": 4.810225800611079e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1470.0} {"timestamp_utc": "2026-04-11T22:32:17Z", "mode": "train", "global_step": 1471, "epoch": 0.056804139635464934, "loss": -0.046, "grad_norm": 2.470545768737793, "learning_rate": 5.545454545454546e-06, "num_tokens": 3208869.0, "completions/mean_length": 203.875, "completions/min_length": 159.0, "completions/max_length": 252.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 203.875, "completions/min_terminated_length": 159.0, "completions/max_terminated_length": 252.0, "rewards/meter/mean": 0.12683206796646118, "rewards/meter/std": 0.18160448968410492, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.08908706158399582, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.097513847053051, "rewards/total_composite/std": 0.14978604018688202, "reward": 0.097513847053051, "reward_std": 0.14978602528572083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029468132182955742, "sampling/sampling_logp_difference/max": 3.1992859840393066, "sampling/importance_sampling_ratio/min": 0.04079132154583931, "sampling/importance_sampling_ratio/mean": 1.000878930091858, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11928362492471933, "clip_ratio/low_mean": 0.01356031943578273, "clip_ratio/low_min": 0.01356031943578273, "clip_ratio/high_mean": 0.010399379534646869, "clip_ratio/high_max": 0.010399379534646869, "clip_ratio/region_mean": 0.0239596989704296, "reward_total_mean": 0.097513847053051, "reward_meter_mean": 0.12683206796646118, "reward_meter_std": 0.18160448968410492, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.08908706158399582, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.097513847053051, "reward_total_composite_std": 0.14978604018688202, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1471.0} {"timestamp_utc": "2026-04-11T22:32:23Z", "mode": "train", "global_step": 1472, "epoch": 0.05684275563793636, "loss": -0.0452, "grad_norm": 1.1331740617752075, "learning_rate": 5.5424242424242425e-06, "num_tokens": 3212086.0, "completions/mean_length": 205.125, "completions/min_length": 188.0, "completions/max_length": 217.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 205.125, "completions/min_terminated_length": 188.0, "completions/max_terminated_length": 217.0, "rewards/meter/mean": 0.9845923185348511, "rewards/meter/std": 0.007074801716953516, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625820279121399, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9226595163345337, "rewards/total_composite/std": 0.0803283229470253, "reward": 0.9226595163345337, "reward_std": 0.0803283154964447, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003899571718648076, "sampling/sampling_logp_difference/max": 0.9814281463623047, "sampling/importance_sampling_ratio/min": 0.5163331627845764, "sampling/importance_sampling_ratio/mean": 1.0015116930007935, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.017492040060460567, "clip_ratio/low_mean": 0.0012930342927575111, "clip_ratio/low_min": 0.0012930342927575111, "clip_ratio/high_mean": 0.001174122968222946, "clip_ratio/high_max": 0.001174122968222946, "clip_ratio/region_mean": 0.002467157260980457, "reward_total_mean": 0.9226595163345337, "reward_meter_mean": 0.9845923185348511, "reward_meter_std": 0.007074801716953516, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625820279121399, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9226595163345337, "reward_total_composite_std": 0.0803283229470253, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1472.0} {"timestamp_utc": "2026-04-11T22:32:28Z", "mode": "train", "global_step": 1473, "epoch": 0.05688137164040778, "loss": -0.0119, "grad_norm": 2.205232858657837, "learning_rate": 5.53939393939394e-06, "num_tokens": 3213623.0, "completions/mean_length": 36.125, "completions/min_length": 35.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9801546335220337, "rewards/meter/std": 0.0024503760505467653, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9801546335220337, "rewards/total_composite/std": 0.0024503760505467653, "reward": 0.9801546335220337, "reward_std": 0.0024503725580871105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009444638155400753, "sampling/sampling_logp_difference/max": 0.6984673738479614, "sampling/importance_sampling_ratio/min": 0.497346967458725, "sampling/importance_sampling_ratio/mean": 0.9962561130523682, "sampling/importance_sampling_ratio/max": 1.189310908317566, "entropy": 0.0205169222317636, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006756756920367479, "clip_ratio/high_max": 0.006756756920367479, "clip_ratio/region_mean": 0.006756756920367479, "reward_total_mean": 0.9801546335220337, "reward_meter_mean": 0.9801546335220337, "reward_meter_std": 0.0024503760505467653, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9801546335220337, "reward_total_composite_std": 0.0024503760505467653, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1473.0} {"timestamp_utc": "2026-04-11T22:32:35Z", "mode": "train", "global_step": 1474, "epoch": 0.056919987642879206, "loss": -0.0437, "grad_norm": 3.7054903507232666, "learning_rate": 5.536363636363636e-06, "num_tokens": 3217368.0, "completions/mean_length": 209.125, "completions/min_length": 197.0, "completions/max_length": 241.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 209.125, "completions/min_terminated_length": 197.0, "completions/max_terminated_length": 241.0, "rewards/meter/mean": 0.9736257791519165, "rewards/meter/std": 0.04814593493938446, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.05050762742757797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8498024344444275, "rewards/total_composite/std": 0.0040647597052156925, "reward": 0.8498024344444275, "reward_std": 0.004064771346747875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006575093138962984, "sampling/sampling_logp_difference/max": 1.4298882484436035, "sampling/importance_sampling_ratio/min": 0.23933567106723785, "sampling/importance_sampling_ratio/mean": 1.0021647214889526, "sampling/importance_sampling_ratio/max": 1.6933517456054688, "entropy": 0.031935454811900854, "clip_ratio/low_mean": 0.004368760972283781, "clip_ratio/low_min": 0.004368760972283781, "clip_ratio/high_mean": 0.0011160714784637094, "clip_ratio/high_max": 0.0011160714784637094, "clip_ratio/region_mean": 0.00548483245074749, "reward_total_mean": 0.8498024344444275, "reward_meter_mean": 0.9736257791519165, "reward_meter_std": 0.04814593493938446, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.05050762742757797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8498024344444275, "reward_total_composite_std": 0.0040647597052156925, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1474.0} {"timestamp_utc": "2026-04-11T22:32:40Z", "mode": "train", "global_step": 1475, "epoch": 0.05695860364535063, "loss": 0.0207, "grad_norm": 3.322364091873169, "learning_rate": 5.533333333333334e-06, "num_tokens": 3219586.0, "completions/mean_length": 126.25, "completions/min_length": 121.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.25, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.19285553693771362, "rewards/meter/std": 0.3013523817062378, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.1285703480243683, "rewards/total_composite/std": 0.200901597738266, "reward": 0.1285703480243683, "reward_std": 0.2009015679359436, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02412356063723564, "sampling/sampling_logp_difference/max": 1.3619365692138672, "sampling/importance_sampling_ratio/min": 0.25616419315338135, "sampling/importance_sampling_ratio/mean": 1.0015215873718262, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10869229305535555, "clip_ratio/low_mean": 0.01259217329788953, "clip_ratio/low_min": 0.01259217329788953, "clip_ratio/high_mean": 0.008181014796718955, "clip_ratio/high_max": 0.008181014796718955, "clip_ratio/region_mean": 0.020773188094608486, "reward_total_mean": 0.1285703480243683, "reward_meter_mean": 0.19285553693771362, "reward_meter_std": 0.3013523817062378, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.1285703480243683, "reward_total_composite_std": 0.200901597738266, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1475.0} {"timestamp_utc": "2026-04-11T22:32:47Z", "mode": "train", "global_step": 1476, "epoch": 0.056997219647822055, "loss": -0.0103, "grad_norm": 1.4927557706832886, "learning_rate": 5.530303030303031e-06, "num_tokens": 3222853.0, "completions/mean_length": 223.375, "completions/min_length": 213.0, "completions/max_length": 235.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 223.375, "completions/min_terminated_length": 213.0, "completions/max_terminated_length": 235.0, "rewards/meter/mean": 0.2282843291759491, "rewards/meter/std": 0.2794889807701111, "rewards/count_adherence/mean": 0.828125, "rewards/count_adherence/std": 0.06469365209341049, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.19736044108867645, "rewards/total_composite/std": 0.2461039423942566, "reward": 0.19736044108867645, "reward_std": 0.2461039423942566, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025218278169631958, "sampling/sampling_logp_difference/max": 1.631584644317627, "sampling/importance_sampling_ratio/min": 0.1956193447113037, "sampling/importance_sampling_ratio/mean": 1.001314401626587, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14154405239969492, "clip_ratio/low_mean": 0.008024285780265927, "clip_ratio/low_min": 0.008024285780265927, "clip_ratio/high_mean": 0.016445244196802378, "clip_ratio/high_max": 0.016445244196802378, "clip_ratio/region_mean": 0.024469529977068305, "reward_total_mean": 0.19736044108867645, "reward_meter_mean": 0.2282843291759491, "reward_meter_std": 0.2794889807701111, "reward_count_adherence_mean": 0.828125, "reward_count_adherence_std": 0.06469365209341049, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.19736044108867645, "reward_total_composite_std": 0.2461039423942566, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1476.0} {"timestamp_utc": "2026-04-11T22:32:53Z", "mode": "train", "global_step": 1477, "epoch": 0.05703583565029348, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.527272727272728e-06, "num_tokens": 3225237.0, "completions/mean_length": 138.0, "completions/min_length": 138.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.0, "completions/min_terminated_length": 138.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9809948205947876, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6539965271949768, "rewards/total_composite/std": 0.0, "reward": 0.6539965271949768, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0003877740236930549, "sampling/sampling_logp_difference/max": 0.03502543270587921, "sampling/importance_sampling_ratio/min": 0.9655808806419373, "sampling/importance_sampling_ratio/mean": 1.0003151893615723, "sampling/importance_sampling_ratio/max": 1.0247117280960083, "entropy": 0.0037024810444563627, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.6539965271949768, "reward_meter_mean": 0.9809948205947876, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6539965271949768, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1477.0} {"timestamp_utc": "2026-04-11T22:33:03Z", "mode": "train", "global_step": 1478, "epoch": 0.0570744516527649, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.524242424242424e-06, "num_tokens": 3226949.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9750425815582275, "rewards/meter/std": 0.039400987327098846, "rewards/count_adherence/mean": 0.8676470518112183, "rewards/count_adherence/std": 0.05214148387312889, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8444418907165527, "rewards/total_composite/std": 0.0273338221013546, "reward": 0.8444418907165527, "reward_std": 0.02733382023870945, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8444418907165527, "reward_meter_mean": 0.9750425815582275, "reward_meter_std": 0.039400987327098846, "reward_count_adherence_mean": 0.8676470518112183, "reward_count_adherence_std": 0.05214148387312889, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8444418907165527, "reward_total_composite_std": 0.0273338221013546, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1478.0} {"timestamp_utc": "2026-04-11T22:33:08Z", "mode": "train", "global_step": 1479, "epoch": 0.05711306765523633, "loss": 0.1159, "grad_norm": 5.19267463684082, "learning_rate": 5.521212121212122e-06, "num_tokens": 3228745.0, "completions/mean_length": 68.5, "completions/min_length": 56.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.5, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.6610715985298157, "rewards/meter/std": 0.24207550287246704, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6372686624526978, "rewards/total_composite/std": 0.27996301651000977, "reward": 0.6372686624526978, "reward_std": 0.2799629867076874, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03689306229352951, "sampling/sampling_logp_difference/max": 1.4248723983764648, "sampling/importance_sampling_ratio/min": 0.24053916335105896, "sampling/importance_sampling_ratio/mean": 1.0003572702407837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16649867966771126, "clip_ratio/low_mean": 0.020222752587869763, "clip_ratio/low_min": 0.020222752587869763, "clip_ratio/high_mean": 0.019961362122558057, "clip_ratio/high_max": 0.019961362122558057, "clip_ratio/region_mean": 0.04018411471042782, "reward_total_mean": 0.6372686624526978, "reward_meter_mean": 0.6610715985298157, "reward_meter_std": 0.24207550287246704, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6372686624526978, "reward_total_composite_std": 0.27996301651000977, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1479.0} {"timestamp_utc": "2026-04-11T22:33:12Z", "mode": "train", "global_step": 1480, "epoch": 0.05715168365770775, "loss": 0.0216, "grad_norm": 9.539521217346191, "learning_rate": 5.518181818181818e-06, "num_tokens": 3230404.0, "completions/mean_length": 56.375, "completions/min_length": 51.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.375, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.7773540019989014, "rewards/meter/std": 0.20132210850715637, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7773540019989014, "rewards/total_composite/std": 0.20132210850715637, "reward": 0.7773540019989014, "reward_std": 0.20132209360599518, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03350376710295677, "sampling/sampling_logp_difference/max": 2.272639274597168, "sampling/importance_sampling_ratio/min": 0.10303986817598343, "sampling/importance_sampling_ratio/mean": 1.006143569946289, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12016534339636564, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/high_mean": 0.01608706102706492, "clip_ratio/high_max": 0.01608706102706492, "clip_ratio/region_mean": 0.020473025972023606, "reward_total_mean": 0.7773540019989014, "reward_meter_mean": 0.7773540019989014, "reward_meter_std": 0.20132210850715637, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7773540019989014, "reward_total_composite_std": 0.20132210850715637, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1480.0} {"timestamp_utc": "2026-04-11T22:33:19Z", "mode": "train", "global_step": 1481, "epoch": 0.057190299660179175, "loss": 0.0321, "grad_norm": 2.034233570098877, "learning_rate": 5.515151515151515e-06, "num_tokens": 3233325.0, "completions/mean_length": 185.125, "completions/min_length": 168.0, "completions/max_length": 196.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 185.125, "completions/min_terminated_length": 168.0, "completions/max_terminated_length": 196.0, "rewards/meter/mean": 0.13062316179275513, "rewards/meter/std": 0.3078117370605469, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.13062316179275513, "rewards/total_composite/std": 0.3078117370605469, "reward": 0.13062316179275513, "reward_std": 0.3078117370605469, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012633098289370537, "sampling/sampling_logp_difference/max": 1.554166555404663, "sampling/importance_sampling_ratio/min": 0.3427262008190155, "sampling/importance_sampling_ratio/mean": 1.0047366619110107, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04980473592877388, "clip_ratio/low_mean": 0.013372937683016062, "clip_ratio/low_min": 0.013372937683016062, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.013372937683016062, "reward_total_mean": 0.13062316179275513, "reward_meter_mean": 0.13062316179275513, "reward_meter_std": 0.3078117370605469, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.13062316179275513, "reward_total_composite_std": 0.3078117370605469, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1481.0} {"timestamp_utc": "2026-04-11T22:33:23Z", "mode": "train", "global_step": 1482, "epoch": 0.0572289156626506, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.512121212121213e-06, "num_tokens": 3235229.0, "completions/mean_length": 73.0, "completions/min_length": 73.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9973368644714355, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9973368644714355, "rewards/total_composite/std": 0.0, "reward": 0.9973368644714355, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00048255512956529856, "sampling/sampling_logp_difference/max": 0.03055346943438053, "sampling/importance_sampling_ratio/min": 0.9699085354804993, "sampling/importance_sampling_ratio/mean": 1.0002455711364746, "sampling/importance_sampling_ratio/max": 1.0242986679077148, "entropy": 0.004239874251652509, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9973368644714355, "reward_meter_mean": 0.9973368644714355, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9973368644714355, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1482.0} {"timestamp_utc": "2026-04-11T22:33:28Z", "mode": "train", "global_step": 1483, "epoch": 0.05726753166512202, "loss": 0.037, "grad_norm": 6.700347900390625, "learning_rate": 5.50909090909091e-06, "num_tokens": 3236855.0, "completions/mean_length": 55.25, "completions/min_length": 48.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8497936725616455, "rewards/meter/std": 0.09216213971376419, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8497936725616455, "rewards/total_composite/std": 0.09216213971376419, "reward": 0.8497936725616455, "reward_std": 0.092162124812603, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02426672726869583, "sampling/sampling_logp_difference/max": 0.8690887093544006, "sampling/importance_sampling_ratio/min": 0.4193335175514221, "sampling/importance_sampling_ratio/mean": 0.9975988268852234, "sampling/importance_sampling_ratio/max": 1.5076543092727661, "entropy": 0.1161250676959753, "clip_ratio/low_mean": 0.01567428675480187, "clip_ratio/low_min": 0.01567428675480187, "clip_ratio/high_mean": 0.006990131689235568, "clip_ratio/high_max": 0.006990131689235568, "clip_ratio/region_mean": 0.022664418444037437, "reward_total_mean": 0.8497936725616455, "reward_meter_mean": 0.8497936725616455, "reward_meter_std": 0.09216213971376419, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8497936725616455, "reward_total_composite_std": 0.09216213971376419, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0, "_plot_step": 1483.0} {"timestamp_utc": "2026-04-11T22:36:55Z", "mode": "train", "global_step": 651, "epoch": 0.026147728642005062, "loss": -0.0071, "grad_norm": 9.100071907043457, "learning_rate": 8.03030303030303e-06, "num_tokens": 1409555.0, "completions/mean_length": 35.125, "completions/min_length": 33.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9935092329978943, "rewards/meter/std": 0.000985646271146834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9935092329978943, "rewards/total_composite/std": 0.000985646271146834, "reward": 0.9935092329978943, "reward_std": 0.0009856420801952481, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04576735198497772, "sampling/sampling_logp_difference/max": 0.9458228349685669, "sampling/importance_sampling_ratio/min": 0.3883599042892456, "sampling/importance_sampling_ratio/mean": 1.011391282081604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17123969458043575, "clip_ratio/low_mean": 0.028315248200669885, "clip_ratio/low_min": 0.028315248200669885, "clip_ratio/high_mean": 0.010521235642954707, "clip_ratio/high_max": 0.010521235642954707, "clip_ratio/region_mean": 0.03883648384362459, "reward_total_mean": 0.9935092329978943, "reward_meter_mean": 0.9935092329978943, "reward_meter_std": 0.000985646271146834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9935092329978943, "reward_total_composite_std": 0.000985646271146834, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1484.0} {"timestamp_utc": "2026-04-11T22:37:05Z", "mode": "train", "global_step": 652, "epoch": 0.026187894123790016, "loss": -0.2894, "grad_norm": 0.4141400456428528, "learning_rate": 8.027272727272728e-06, "num_tokens": 1413749.0, "completions/mean_length": 374.25, "completions/min_length": 343.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 354.5714416503906, "completions/min_terminated_length": 343.0, "completions/max_terminated_length": 363.0, "rewards/meter/mean": 0.8732885122299194, "rewards/meter/std": 0.3528631627559662, "rewards/count_adherence/mean": 0.824999988079071, "rewards/count_adherence/std": 0.337003618478775, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.17251461744308472, "rewards/repeat_penalty/std": 0.33435773849487305, "rewards/total_composite/mean": 0.04464861750602722, "rewards/total_composite/std": 0.018085261806845665, "reward": 0.04464861750602722, "reward_std": 0.018085261806845665, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004473666660487652, "sampling/sampling_logp_difference/max": 0.7091238498687744, "sampling/importance_sampling_ratio/min": 0.4920751452445984, "sampling/importance_sampling_ratio/mean": 1.0013567209243774, "sampling/importance_sampling_ratio/max": 1.834702491760254, "entropy": 0.019337893230840564, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0035446555411908776, "clip_ratio/high_max": 0.0035446555411908776, "clip_ratio/region_mean": 0.0035446555411908776, "reward_total_mean": 0.04464861750602722, "reward_meter_mean": 0.8732885122299194, "reward_meter_std": 0.3528631627559662, "reward_count_adherence_mean": 0.824999988079071, "reward_count_adherence_std": 0.337003618478775, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.17251461744308472, "reward_repeat_penalty_std": 0.33435773849487305, "reward_total_composite_mean": 0.04464861750602722, "reward_total_composite_std": 0.018085261806845665, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1485.0} {"timestamp_utc": "2026-04-11T22:37:15Z", "mode": "train", "global_step": 653, "epoch": 0.02622805960557497, "loss": 0.1416, "grad_norm": 1.191830039024353, "learning_rate": 8.024242424242425e-06, "num_tokens": 1415617.0, "completions/mean_length": 133.5, "completions/min_length": 78.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 79.42857360839844, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9687988758087158, "rewards/meter/std": 0.06074121594429016, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4583333432674408, "rewards/repeat_penalty/std": 0.24800792336463928, "rewards/total_composite/mean": 0.43250638246536255, "rewards/total_composite/std": 0.1944577693939209, "reward": 0.43250638246536255, "reward_std": 0.1944577544927597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014217361807823181, "sampling/sampling_logp_difference/max": 0.9473090171813965, "sampling/importance_sampling_ratio/min": 0.3877831697463989, "sampling/importance_sampling_ratio/mean": 1.0012526512145996, "sampling/importance_sampling_ratio/max": 1.5146371126174927, "entropy": 0.0790142323821783, "clip_ratio/low_mean": 0.007873826543800533, "clip_ratio/low_min": 0.007873826543800533, "clip_ratio/high_mean": 0.0029761905316263437, "clip_ratio/high_max": 0.0029761905316263437, "clip_ratio/region_mean": 0.010850017075426877, "reward_total_mean": 0.43250638246536255, "reward_meter_mean": 0.9687988758087158, "reward_meter_std": 0.06074121594429016, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4583333432674408, "reward_repeat_penalty_std": 0.24800792336463928, "reward_total_composite_mean": 0.43250638246536255, "reward_total_composite_std": 0.1944577693939209, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1486.0} {"timestamp_utc": "2026-04-11T22:37:21Z", "mode": "train", "global_step": 654, "epoch": 0.026268225087359924, "loss": -0.0232, "grad_norm": 1.927111268043518, "learning_rate": 8.021212121212122e-06, "num_tokens": 1418478.0, "completions/mean_length": 170.625, "completions/min_length": 162.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.625, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9953961968421936, "rewards/meter/std": 0.0023201238363981247, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.2678571343421936, "rewards/repeat_penalty/std": 0.17806050181388855, "rewards/total_composite/mean": 0.2665513753890991, "rewards/total_composite/std": 0.17725923657417297, "reward": 0.2665513753890991, "reward_std": 0.17725922167301178, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01804671622812748, "sampling/sampling_logp_difference/max": 1.4201292991638184, "sampling/importance_sampling_ratio/min": 0.24168278276920319, "sampling/importance_sampling_ratio/mean": 1.0009305477142334, "sampling/importance_sampling_ratio/max": 1.6224781274795532, "entropy": 0.08471588138490915, "clip_ratio/low_mean": 0.003801907878369093, "clip_ratio/low_min": 0.003801907878369093, "clip_ratio/high_mean": 0.008620842476375401, "clip_ratio/high_max": 0.008620842476375401, "clip_ratio/region_mean": 0.012422750354744494, "reward_total_mean": 0.2665513753890991, "reward_meter_mean": 0.9953961968421936, "reward_meter_std": 0.0023201238363981247, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.2678571343421936, "reward_repeat_penalty_std": 0.17806050181388855, "reward_total_composite_mean": 0.2665513753890991, "reward_total_composite_std": 0.17725923657417297, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1487.0} {"timestamp_utc": "2026-04-11T22:37:30Z", "mode": "train", "global_step": 655, "epoch": 0.026308390569144878, "loss": -0.0945, "grad_norm": 2.2248940467834473, "learning_rate": 8.018181818181818e-06, "num_tokens": 1420163.0, "completions/mean_length": 116.625, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 60.142860412597656, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7503823637962341, "rewards/meter/std": 0.44998785853385925, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.30860671401023865, "rewards/total_composite/mean": 0.25364238023757935, "rewards/total_composite/std": 0.14388985931873322, "reward": 0.25364238023757935, "reward_std": 0.1438898742198944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024109482765197754, "sampling/sampling_logp_difference/max": 1.0410680770874023, "sampling/importance_sampling_ratio/min": 0.35307735204696655, "sampling/importance_sampling_ratio/mean": 1.0044450759887695, "sampling/importance_sampling_ratio/max": 1.6566717624664307, "entropy": 0.17866009753197432, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/high_mean": 0.012714299838989973, "clip_ratio/high_max": 0.012714299838989973, "clip_ratio/region_mean": 0.01455253513995558, "reward_total_mean": 0.25364238023757935, "reward_meter_mean": 0.7503823637962341, "reward_meter_std": 0.44998785853385925, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.30860671401023865, "reward_total_composite_mean": 0.25364238023757935, "reward_total_composite_std": 0.14388985931873322, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1488.0} {"timestamp_utc": "2026-04-11T22:37:35Z", "mode": "train", "global_step": 656, "epoch": 0.026348556050929832, "loss": 0.0094, "grad_norm": 6.352079391479492, "learning_rate": 8.015151515151515e-06, "num_tokens": 1421891.0, "completions/mean_length": 58.0, "completions/min_length": 55.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9239930510520935, "rewards/meter/std": 0.15441425144672394, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.3333333432674408, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.3079977035522461, "rewards/total_composite/std": 0.051471415907144547, "reward": 0.3079977035522461, "reward_std": 0.05147142335772514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022419268265366554, "sampling/sampling_logp_difference/max": 0.8770105838775635, "sampling/importance_sampling_ratio/min": 0.41602474451065063, "sampling/importance_sampling_ratio/mean": 1.0036613941192627, "sampling/importance_sampling_ratio/max": 1.928532361984253, "entropy": 0.16807558294385672, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.01278492109850049, "clip_ratio/high_max": 0.01278492109850049, "clip_ratio/region_mean": 0.01709526591002941, "reward_total_mean": 0.3079977035522461, "reward_meter_mean": 0.9239930510520935, "reward_meter_std": 0.15441425144672394, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.3333333432674408, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.3079977035522461, "reward_total_composite_std": 0.051471415907144547, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1489.0} {"timestamp_utc": "2026-04-11T22:37:41Z", "mode": "train", "global_step": 657, "epoch": 0.026388721532714786, "loss": 0.0154, "grad_norm": 2.9878616333007812, "learning_rate": 8.012121212121214e-06, "num_tokens": 1424909.0, "completions/mean_length": 181.25, "completions/min_length": 175.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.25, "completions/min_terminated_length": 175.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.9874005317687988, "rewards/meter/std": 0.011436098255217075, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.20192307233810425, "rewards/repeat_penalty/std": 0.2094048410654068, "rewards/total_composite/mean": 0.1662987768650055, "rewards/total_composite/std": 0.17312301695346832, "reward": 0.1662987768650055, "reward_std": 0.17312301695346832, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010927603580057621, "sampling/sampling_logp_difference/max": 1.7186641693115234, "sampling/importance_sampling_ratio/min": 0.1793055236339569, "sampling/importance_sampling_ratio/mean": 1.0016026496887207, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.042974324664101005, "clip_ratio/low_mean": 0.008925562433432788, "clip_ratio/low_min": 0.008925562433432788, "clip_ratio/high_mean": 0.0028169237775728106, "clip_ratio/high_max": 0.0028169237775728106, "clip_ratio/region_mean": 0.011742486211005598, "reward_total_mean": 0.1662987768650055, "reward_meter_mean": 0.9874005317687988, "reward_meter_std": 0.011436098255217075, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.20192307233810425, "reward_repeat_penalty_std": 0.2094048410654068, "reward_total_composite_mean": 0.1662987768650055, "reward_total_composite_std": 0.17312301695346832, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1490.0} {"timestamp_utc": "2026-04-11T22:37:51Z", "mode": "train", "global_step": 658, "epoch": 0.02642888701449974, "loss": -0.0762, "grad_norm": 0.7183297276496887, "learning_rate": 8.00909090909091e-06, "num_tokens": 1429785.0, "completions/mean_length": 428.5, "completions/min_length": 394.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 416.5714416503906, "completions/min_terminated_length": 394.0, "completions/max_terminated_length": 441.0, "rewards/meter/mean": 0.8735865354537964, "rewards/meter/std": 0.35298284888267517, "rewards/count_adherence/mean": 0.6416666507720947, "rewards/count_adherence/std": 0.26170989871025085, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.17413419485092163, "rewards/repeat_penalty/std": 0.34882497787475586, "rewards/total_composite/mean": 0.033445145934820175, "rewards/total_composite/std": 0.06880706548690796, "reward": 0.033445145934820175, "reward_std": 0.06880706548690796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0026275559794157743, "sampling/sampling_logp_difference/max": 1.0090997219085693, "sampling/importance_sampling_ratio/min": 0.3645470440387726, "sampling/importance_sampling_ratio/mean": 1.0009920597076416, "sampling/importance_sampling_ratio/max": 1.7735869884490967, "entropy": 0.012874894309788942, "clip_ratio/low_mean": 0.00030637255986221135, "clip_ratio/low_min": 0.00030637255986221135, "clip_ratio/high_mean": 0.0005817760829813778, "clip_ratio/high_max": 0.0005817760829813778, "clip_ratio/region_mean": 0.0008881486428435892, "reward_total_mean": 0.033445145934820175, "reward_meter_mean": 0.8735865354537964, "reward_meter_std": 0.35298284888267517, "reward_count_adherence_mean": 0.6416666507720947, "reward_count_adherence_std": 0.26170989871025085, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.17413419485092163, "reward_repeat_penalty_std": 0.34882497787475586, "reward_total_composite_mean": 0.033445145934820175, "reward_total_composite_std": 0.06880706548690796, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1491.0} {"timestamp_utc": "2026-04-11T22:38:00Z", "mode": "train", "global_step": 659, "epoch": 0.026469052496284694, "loss": -0.1652, "grad_norm": 1.3609012365341187, "learning_rate": 8.006060606060607e-06, "num_tokens": 1431540.0, "completions/mean_length": 190.375, "completions/min_length": 81.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 83.16667175292969, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9505882263183594, "rewards/meter/std": 0.030698692426085472, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.513268232345581, "rewards/total_composite/std": 0.3378344178199768, "reward": 0.513268232345581, "reward_std": 0.3378343880176544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02022649347782135, "sampling/sampling_logp_difference/max": 1.8950445652008057, "sampling/importance_sampling_ratio/min": 0.1503116339445114, "sampling/importance_sampling_ratio/mean": 1.0079931020736694, "sampling/importance_sampling_ratio/max": 1.7507625818252563, "entropy": 0.06744233565405011, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.008969587041065097, "clip_ratio/high_max": 0.008969587041065097, "clip_ratio/region_mean": 0.008969587041065097, "reward_total_mean": 0.513268232345581, "reward_meter_mean": 0.9505882263183594, "reward_meter_std": 0.030698692426085472, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.513268232345581, "reward_total_composite_std": 0.3378344178199768, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1492.0} {"timestamp_utc": "2026-04-11T22:38:05Z", "mode": "train", "global_step": 660, "epoch": 0.026509217978069648, "loss": 0.0308, "grad_norm": 2.2579872608184814, "learning_rate": 8.003030303030304e-06, "num_tokens": 1433681.0, "completions/mean_length": 110.625, "completions/min_length": 104.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.625, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9927444458007812, "rewards/meter/std": 0.0012347318697720766, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.23571428656578064, "rewards/repeat_penalty/std": 0.07284314185380936, "rewards/total_composite/mean": 0.22219520807266235, "rewards/total_composite/std": 0.07082200050354004, "reward": 0.22219520807266235, "reward_std": 0.07082199305295944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008687403053045273, "sampling/sampling_logp_difference/max": 1.8446245193481445, "sampling/importance_sampling_ratio/min": 0.1580846756696701, "sampling/importance_sampling_ratio/mean": 1.001028299331665, "sampling/importance_sampling_ratio/max": 1.6747108697891235, "entropy": 0.020129066659137607, "clip_ratio/low_mean": 0.006200430449098349, "clip_ratio/low_min": 0.006200430449098349, "clip_ratio/high_mean": 0.0036057692486792803, "clip_ratio/high_max": 0.0036057692486792803, "clip_ratio/region_mean": 0.009806199697777629, "reward_total_mean": 0.22219520807266235, "reward_meter_mean": 0.9927444458007812, "reward_meter_std": 0.0012347318697720766, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.23571428656578064, "reward_repeat_penalty_std": 0.07284314185380936, "reward_total_composite_mean": 0.22219520807266235, "reward_total_composite_std": 0.07082200050354004, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1493.0} {"timestamp_utc": "2026-04-11T22:38:15Z", "mode": "train", "global_step": 661, "epoch": 0.0265493834598546, "loss": -0.0515, "grad_norm": 3.181408405303955, "learning_rate": 8.000000000000001e-06, "num_tokens": 1435750.0, "completions/mean_length": 203.625, "completions/min_length": 81.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 100.83333587646484, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.43306416273117065, "rewards/meter/std": 0.44791871309280396, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.6357142925262451, "rewards/repeat_penalty/std": 0.31916436553001404, "rewards/total_composite/mean": 0.22007080912590027, "rewards/total_composite/std": 0.2968771755695343, "reward": 0.22007080912590027, "reward_std": 0.2968771457672119, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04375706985592842, "sampling/sampling_logp_difference/max": 1.372495174407959, "sampling/importance_sampling_ratio/min": 0.253473699092865, "sampling/importance_sampling_ratio/mean": 1.0068386793136597, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17050567921251059, "clip_ratio/low_mean": 0.008919409476220608, "clip_ratio/low_min": 0.008919409476220608, "clip_ratio/high_mean": 0.017718179151415825, "clip_ratio/high_max": 0.017718179151415825, "clip_ratio/region_mean": 0.026637588627636433, "reward_total_mean": 0.22007080912590027, "reward_meter_mean": 0.43306416273117065, "reward_meter_std": 0.44791871309280396, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.18600596487522125, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.6357142925262451, "reward_repeat_penalty_std": 0.31916436553001404, "reward_total_composite_mean": 0.22007080912590027, "reward_total_composite_std": 0.2968771755695343, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1494.0} {"timestamp_utc": "2026-04-11T22:38:25Z", "mode": "train", "global_step": 662, "epoch": 0.026589548941639556, "loss": -0.1279, "grad_norm": 2.6338069438934326, "learning_rate": 7.996969696969697e-06, "num_tokens": 1437638.0, "completions/mean_length": 117.0, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 60.57143020629883, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9266777634620667, "rewards/meter/std": 0.1883465051651001, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5833333730697632, "rewards/repeat_penalty/std": 0.2357022762298584, "rewards/total_composite/mean": 0.455108642578125, "rewards/total_composite/std": 0.24599777162075043, "reward": 0.455108642578125, "reward_std": 0.24599777162075043, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032812319695949554, "sampling/sampling_logp_difference/max": 1.3324346542358398, "sampling/importance_sampling_ratio/min": 0.263834148645401, "sampling/importance_sampling_ratio/mean": 1.0021228790283203, "sampling/importance_sampling_ratio/max": 1.474334478378296, "entropy": 0.17267457023262978, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/high_mean": 0.012073024990968406, "clip_ratio/high_max": 0.012073024990968406, "clip_ratio/region_mean": 0.016458989935927093, "reward_total_mean": 0.455108642578125, "reward_meter_mean": 0.9266777634620667, "reward_meter_std": 0.1883465051651001, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5833333730697632, "reward_repeat_penalty_std": 0.2357022762298584, "reward_total_composite_mean": 0.455108642578125, "reward_total_composite_std": 0.24599777162075043, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1495.0} {"timestamp_utc": "2026-04-11T22:38:30Z", "mode": "train", "global_step": 663, "epoch": 0.02662971442342451, "loss": 0.021, "grad_norm": 4.197593688964844, "learning_rate": 7.993939393939396e-06, "num_tokens": 1439950.0, "completions/mean_length": 108.0, "completions/min_length": 96.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9684537053108215, "rewards/meter/std": 0.027583837509155273, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4226190447807312, "rewards/repeat_penalty/std": 0.2210753709077835, "rewards/total_composite/mean": 0.3900730013847351, "rewards/total_composite/std": 0.19891957938671112, "reward": 0.3900730013847351, "reward_std": 0.19891956448554993, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02085985243320465, "sampling/sampling_logp_difference/max": 1.0295929908752441, "sampling/importance_sampling_ratio/min": 0.3571523129940033, "sampling/importance_sampling_ratio/mean": 0.9997091889381409, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11175651382654905, "clip_ratio/low_mean": 0.00461137923412025, "clip_ratio/low_min": 0.00461137923412025, "clip_ratio/high_mean": 0.014703674940392375, "clip_ratio/high_max": 0.014703674940392375, "clip_ratio/region_mean": 0.019315054174512625, "reward_total_mean": 0.3900730013847351, "reward_meter_mean": 0.9684537053108215, "reward_meter_std": 0.027583837509155273, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4226190447807312, "reward_repeat_penalty_std": 0.2210753709077835, "reward_total_composite_mean": 0.3900730013847351, "reward_total_composite_std": 0.19891957938671112, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1496.0} {"timestamp_utc": "2026-04-11T22:38:35Z", "mode": "train", "global_step": 664, "epoch": 0.026669879905209463, "loss": -0.029, "grad_norm": 7.99793004989624, "learning_rate": 7.990909090909091e-06, "num_tokens": 1441763.0, "completions/mean_length": 70.625, "completions/min_length": 68.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.625, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9828510880470276, "rewards/meter/std": 0.025825461372733116, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5833333730697632, "rewards/repeat_penalty/std": 0.29546841979026794, "rewards/total_composite/mean": 0.568142294883728, "rewards/total_composite/std": 0.27455055713653564, "reward": 0.568142294883728, "reward_std": 0.27455058693885803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03234035521745682, "sampling/sampling_logp_difference/max": 1.1230463981628418, "sampling/importance_sampling_ratio/min": 0.3252873122692108, "sampling/importance_sampling_ratio/mean": 1.0028775930404663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10323597816750407, "clip_ratio/low_mean": 0.0055147059028968215, "clip_ratio/low_min": 0.0055147059028968215, "clip_ratio/high_mean": 0.023867564275860786, "clip_ratio/high_max": 0.023867564275860786, "clip_ratio/region_mean": 0.029382270178757608, "reward_total_mean": 0.568142294883728, "reward_meter_mean": 0.9828510880470276, "reward_meter_std": 0.025825461372733116, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5833333730697632, "reward_repeat_penalty_std": 0.29546841979026794, "reward_total_composite_mean": 0.568142294883728, "reward_total_composite_std": 0.27455055713653564, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1497.0} {"timestamp_utc": "2026-04-11T22:38:39Z", "mode": "train", "global_step": 665, "epoch": 0.026710045386994417, "loss": -0.0097, "grad_norm": 3.2300848960876465, "learning_rate": 7.987878787878789e-06, "num_tokens": 1443616.0, "completions/mean_length": 80.625, "completions/min_length": 74.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.625, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9603564143180847, "rewards/meter/std": 0.023673059418797493, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7601395845413208, "rewards/total_composite/std": 0.16597026586532593, "reward": 0.7601395845413208, "reward_std": 0.16597026586532593, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0124077582731843, "sampling/sampling_logp_difference/max": 1.0308361053466797, "sampling/importance_sampling_ratio/min": 0.3567086160182953, "sampling/importance_sampling_ratio/mean": 1.0014386177062988, "sampling/importance_sampling_ratio/max": 1.564980387687683, "entropy": 0.07347342604771256, "clip_ratio/low_mean": 0.009405238670296967, "clip_ratio/low_min": 0.009405238670296967, "clip_ratio/high_mean": 0.0015432098880410194, "clip_ratio/high_max": 0.0015432098880410194, "clip_ratio/region_mean": 0.010948448558337986, "reward_total_mean": 0.7601395845413208, "reward_meter_mean": 0.9603564143180847, "reward_meter_std": 0.023673059418797493, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.7601395845413208, "reward_total_composite_std": 0.16597026586532593, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1498.0} {"timestamp_utc": "2026-04-11T22:38:44Z", "mode": "train", "global_step": 666, "epoch": 0.02675021086877937, "loss": 0.0007, "grad_norm": 3.160210609436035, "learning_rate": 7.984848484848486e-06, "num_tokens": 1445930.0, "completions/mean_length": 122.25, "completions/min_length": 120.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.25, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.8794082999229431, "rewards/meter/std": 0.32782238721847534, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.212053582072258, "rewards/repeat_penalty/std": 0.2030286192893982, "rewards/total_composite/mean": 0.12827160954475403, "rewards/total_composite/std": 0.03277355059981346, "reward": 0.12827160954475403, "reward_std": 0.03277355059981346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02072002924978733, "sampling/sampling_logp_difference/max": 6.419788360595703, "sampling/importance_sampling_ratio/min": 0.0016290009953081608, "sampling/importance_sampling_ratio/mean": 0.9980396032333374, "sampling/importance_sampling_ratio/max": 1.6665557622909546, "entropy": 0.03877314692363143, "clip_ratio/low_mean": 0.0010416667209938169, "clip_ratio/low_min": 0.0010416667209938169, "clip_ratio/high_mean": 0.006232782383449376, "clip_ratio/high_max": 0.006232782383449376, "clip_ratio/region_mean": 0.0072744491044431925, "reward_total_mean": 0.12827160954475403, "reward_meter_mean": 0.8794082999229431, "reward_meter_std": 0.32782238721847534, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.212053582072258, "reward_repeat_penalty_std": 0.2030286192893982, "reward_total_composite_mean": 0.12827160954475403, "reward_total_composite_std": 0.03277355059981346, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1499.0} {"timestamp_utc": "2026-04-11T22:38:50Z", "mode": "train", "global_step": 667, "epoch": 0.026790376350564325, "loss": 0.0108, "grad_norm": 4.775012016296387, "learning_rate": 7.981818181818183e-06, "num_tokens": 1448313.0, "completions/mean_length": 120.875, "completions/min_length": 116.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.875, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9965612888336182, "rewards/meter/std": 0.0018690497381612659, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.25, "rewards/repeat_penalty/std": 0.21257823705673218, "rewards/total_composite/mean": 0.24911293387413025, "rewards/total_composite/std": 0.2116239219903946, "reward": 0.24911293387413025, "reward_std": 0.2116239219903946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017011119052767754, "sampling/sampling_logp_difference/max": 1.0500693321228027, "sampling/importance_sampling_ratio/min": 0.3773258924484253, "sampling/importance_sampling_ratio/mean": 0.9989076256752014, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07529849279671907, "clip_ratio/low_mean": 0.0031968391267582774, "clip_ratio/low_min": 0.0031968391267582774, "clip_ratio/high_mean": 0.006372183095663786, "clip_ratio/high_max": 0.006372183095663786, "clip_ratio/region_mean": 0.009569022222422063, "reward_total_mean": 0.24911293387413025, "reward_meter_mean": 0.9965612888336182, "reward_meter_std": 0.0018690497381612659, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.25, "reward_repeat_penalty_std": 0.21257823705673218, "reward_total_composite_mean": 0.24911293387413025, "reward_total_composite_std": 0.2116239219903946, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1500.0} {"timestamp_utc": "2026-04-11T22:38:59Z", "mode": "train", "global_step": 668, "epoch": 0.02683054183234928, "loss": -0.1559, "grad_norm": 2.0396084785461426, "learning_rate": 7.978787878787879e-06, "num_tokens": 1451329.0, "completions/mean_length": 243.0, "completions/min_length": 195.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 204.57144165039062, "completions/min_terminated_length": 195.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.9733182787895203, "rewards/meter/std": 0.02887692302465439, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.27272728085517883, "rewards/repeat_penalty/std": 0.13744163513183594, "rewards/total_composite/mean": 0.17570531368255615, "rewards/total_composite/std": 0.11893084645271301, "reward": 0.17570531368255615, "reward_std": 0.11893083900213242, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01857808604836464, "sampling/sampling_logp_difference/max": 2.0781538486480713, "sampling/importance_sampling_ratio/min": 0.1251610666513443, "sampling/importance_sampling_ratio/mean": 0.9996070861816406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05400602100417018, "clip_ratio/low_mean": 0.007991038146428764, "clip_ratio/low_min": 0.007991038146428764, "clip_ratio/high_mean": 0.0030251864809542894, "clip_ratio/high_max": 0.0030251864809542894, "clip_ratio/region_mean": 0.011016224627383053, "reward_total_mean": 0.17570531368255615, "reward_meter_mean": 0.9733182787895203, "reward_meter_std": 0.02887692302465439, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.27272728085517883, "reward_repeat_penalty_std": 0.13744163513183594, "reward_total_composite_mean": 0.17570531368255615, "reward_total_composite_std": 0.11893084645271301, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1501.0} {"timestamp_utc": "2026-04-11T22:39:09Z", "mode": "train", "global_step": 669, "epoch": 0.026870707314134233, "loss": -0.023, "grad_norm": 2.1599395275115967, "learning_rate": 7.975757575757576e-06, "num_tokens": 1453130.0, "completions/mean_length": 240.125, "completions/min_length": 74.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 77.0, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.5011062026023865, "rewards/meter/std": 0.4051038920879364, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4261804223060608, "rewards/total_composite/std": 0.4667132496833801, "reward": 0.4261804223060608, "reward_std": 0.46671321988105774, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.052589599043130875, "sampling/sampling_logp_difference/max": 1.824247121810913, "sampling/importance_sampling_ratio/min": 0.16133907437324524, "sampling/importance_sampling_ratio/mean": 1.0080572366714478, "sampling/importance_sampling_ratio/max": 1.8760875463485718, "entropy": 0.16993126086890697, "clip_ratio/low_mean": 0.009377967799082398, "clip_ratio/low_min": 0.009377967799082398, "clip_ratio/high_mean": 0.011646514758467674, "clip_ratio/high_max": 0.011646514758467674, "clip_ratio/region_mean": 0.021024482557550073, "reward_total_mean": 0.4261804223060608, "reward_meter_mean": 0.5011062026023865, "reward_meter_std": 0.4051038920879364, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4261804223060608, "reward_total_composite_std": 0.4667132496833801, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1502.0} {"timestamp_utc": "2026-04-11T22:39:18Z", "mode": "train", "global_step": 670, "epoch": 0.026910872795919187, "loss": -0.1072, "grad_norm": 0.766589879989624, "learning_rate": 7.972727272727273e-06, "num_tokens": 1454698.0, "completions/mean_length": 351.0, "completions/min_length": 73.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 82.66667175292969, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.6174333691596985, "rewards/meter/std": 0.4055008888244629, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.375, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.249087393283844, "rewards/total_composite/std": 0.3437734544277191, "reward": 0.249087393283844, "reward_std": 0.3437734842300415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03491988405585289, "sampling/sampling_logp_difference/max": 1.1255141496658325, "sampling/importance_sampling_ratio/min": 0.4397118091583252, "sampling/importance_sampling_ratio/mean": 1.004920482635498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09451978467404842, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.013186233583837748, "clip_ratio/high_max": 0.013186233583837748, "clip_ratio/region_mean": 0.013186233583837748, "reward_total_mean": 0.249087393283844, "reward_meter_mean": 0.6174333691596985, "reward_meter_std": 0.4055008888244629, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.375, "reward_arabic_clean_std": 0.5175492167472839, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_total_composite_mean": 0.249087393283844, "reward_total_composite_std": 0.3437734544277191, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1503.0} {"timestamp_utc": "2026-04-11T22:39:28Z", "mode": "train", "global_step": 671, "epoch": 0.02695103827770414, "loss": -0.0912, "grad_norm": 0.7819209098815918, "learning_rate": 7.96969696969697e-06, "num_tokens": 1456328.0, "completions/mean_length": 354.75, "completions/min_length": 91.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 92.66667175292969, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.7872094511985779, "rewards/meter/std": 0.3365079164505005, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/repeat_penalty/mean": 0.4750000238418579, "rewards/repeat_penalty/std": 0.2121320366859436, "rewards/total_composite/mean": 0.19879300892353058, "rewards/total_composite/std": 0.21251867711544037, "reward": 0.19879300892353058, "reward_std": 0.21251867711544037, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0315895713865757, "sampling/sampling_logp_difference/max": 1.163419246673584, "sampling/importance_sampling_ratio/min": 0.31241610646247864, "sampling/importance_sampling_ratio/mean": 1.0077892541885376, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07205967605113983, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006735671544447541, "clip_ratio/high_max": 0.006735671544447541, "clip_ratio/region_mean": 0.006735671544447541, "reward_total_mean": 0.19879300892353058, "reward_meter_mean": 0.7872094511985779, "reward_meter_std": 0.3365079164505005, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_repeat_penalty_mean": 0.4750000238418579, "reward_repeat_penalty_std": 0.2121320366859436, "reward_total_composite_mean": 0.19879300892353058, "reward_total_composite_std": 0.21251867711544037, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1504.0} {"timestamp_utc": "2026-04-11T22:39:38Z", "mode": "train", "global_step": 672, "epoch": 0.026991203759489095, "loss": -0.1237, "grad_norm": 0.8036666512489319, "learning_rate": 7.966666666666668e-06, "num_tokens": 1458728.0, "completions/mean_length": 314.0, "completions/min_length": 160.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 195.1999969482422, "completions/min_terminated_length": 160.0, "completions/max_terminated_length": 220.0, "rewards/meter/mean": 0.7117201685905457, "rewards/meter/std": 0.3770325481891632, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.1414213627576828, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.5681818127632141, "rewards/repeat_penalty/std": 0.2368127554655075, "rewards/total_composite/mean": 0.23602712154388428, "rewards/total_composite/std": 0.1749935895204544, "reward": 0.23602712154388428, "reward_std": 0.1749935895204544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020951304584741592, "sampling/sampling_logp_difference/max": 1.4866762161254883, "sampling/importance_sampling_ratio/min": 0.22612299025058746, "sampling/importance_sampling_ratio/mean": 0.9994598627090454, "sampling/importance_sampling_ratio/max": 1.857919692993164, "entropy": 0.06148350611329079, "clip_ratio/low_mean": 0.0028696630615741014, "clip_ratio/low_min": 0.0028696630615741014, "clip_ratio/high_mean": 0.010866477387025952, "clip_ratio/high_max": 0.010866477387025952, "clip_ratio/region_mean": 0.013736140448600054, "reward_total_mean": 0.23602712154388428, "reward_meter_mean": 0.7117201685905457, "reward_meter_std": 0.3770325481891632, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.1414213627576828, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.5681818127632141, "reward_repeat_penalty_std": 0.2368127554655075, "reward_total_composite_mean": 0.23602712154388428, "reward_total_composite_std": 0.1749935895204544, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1505.0} {"timestamp_utc": "2026-04-11T22:39:48Z", "mode": "train", "global_step": 673, "epoch": 0.02703136924127405, "loss": -0.0271, "grad_norm": 4.668978214263916, "learning_rate": 7.963636363636365e-06, "num_tokens": 1460405.0, "completions/mean_length": 117.625, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 61.28571701049805, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8629885911941528, "rewards/meter/std": 0.28332778811454773, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.768899142742157, "rewards/total_composite/std": 0.3212806284427643, "reward": 0.768899142742157, "reward_std": 0.32128065824508667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04510435834527016, "sampling/sampling_logp_difference/max": 1.6506894826889038, "sampling/importance_sampling_ratio/min": 0.1919175386428833, "sampling/importance_sampling_ratio/mean": 1.0077714920043945, "sampling/importance_sampling_ratio/max": 1.7451122999191284, "entropy": 0.2579981219023466, "clip_ratio/low_mean": 0.020833334419876337, "clip_ratio/low_min": 0.020833334419876337, "clip_ratio/high_mean": 0.02129602595232427, "clip_ratio/high_max": 0.02129602595232427, "clip_ratio/region_mean": 0.04212936037220061, "reward_total_mean": 0.768899142742157, "reward_meter_mean": 0.8629885911941528, "reward_meter_std": 0.28332778811454773, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.768899142742157, "reward_total_composite_std": 0.3212806284427643, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1506.0} {"timestamp_utc": "2026-04-11T22:39:58Z", "mode": "train", "global_step": 674, "epoch": 0.027071534723059003, "loss": -0.0956, "grad_norm": 3.691714286804199, "learning_rate": 7.96060606060606e-06, "num_tokens": 1462244.0, "completions/mean_length": 128.875, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 74.14286041259766, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.7070336937904358, "rewards/meter/std": 0.39935943484306335, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.5691444873809814, "rewards/total_composite/std": 0.343730628490448, "reward": 0.5691444873809814, "reward_std": 0.3437305986881256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05939958617091179, "sampling/sampling_logp_difference/max": 1.2720730304718018, "sampling/importance_sampling_ratio/min": 0.28025004267692566, "sampling/importance_sampling_ratio/mean": 1.0029557943344116, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23062355443835258, "clip_ratio/low_mean": 0.009934040834195912, "clip_ratio/low_min": 0.009934040834195912, "clip_ratio/high_mean": 0.04135313397273421, "clip_ratio/high_max": 0.04135313397273421, "clip_ratio/region_mean": 0.051287174806930125, "reward_total_mean": 0.5691444873809814, "reward_meter_mean": 0.7070336937904358, "reward_meter_std": 0.39935943484306335, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.5691444873809814, "reward_total_composite_std": 0.343730628490448, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1507.0} {"timestamp_utc": "2026-04-11T22:40:08Z", "mode": "train", "global_step": 675, "epoch": 0.027111700204843957, "loss": -0.0973, "grad_norm": 0.7685384154319763, "learning_rate": 7.957575757575758e-06, "num_tokens": 1466421.0, "completions/mean_length": 376.125, "completions/min_length": 337.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 356.71429443359375, "completions/min_terminated_length": 337.0, "completions/max_terminated_length": 367.0, "rewards/meter/mean": 0.9768602848052979, "rewards/meter/std": 0.0202656090259552, "rewards/count_adherence/mean": 0.8624999523162842, "rewards/count_adherence/std": 0.31139087677001953, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.2991071343421936, "rewards/repeat_penalty/std": 0.36836329102516174, "rewards/total_composite/mean": 0.15836204588413239, "rewards/total_composite/std": 0.21933622658252716, "reward": 0.15836204588413239, "reward_std": 0.21933621168136597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005116640590131283, "sampling/sampling_logp_difference/max": 1.467843770980835, "sampling/importance_sampling_ratio/min": 0.4014633893966675, "sampling/importance_sampling_ratio/mean": 1.0009857416152954, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0205289286095649, "clip_ratio/low_mean": 0.0032165506563615054, "clip_ratio/low_min": 0.0032165506563615054, "clip_ratio/high_mean": 0.000681198900565505, "clip_ratio/high_max": 0.000681198900565505, "clip_ratio/region_mean": 0.0038977495569270104, "reward_total_mean": 0.15836204588413239, "reward_meter_mean": 0.9768602848052979, "reward_meter_std": 0.0202656090259552, "reward_count_adherence_mean": 0.8624999523162842, "reward_count_adherence_std": 0.31139087677001953, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.2991071343421936, "reward_repeat_penalty_std": 0.36836329102516174, "reward_total_composite_mean": 0.15836204588413239, "reward_total_composite_std": 0.21933622658252716, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1508.0} {"timestamp_utc": "2026-04-11T22:40:12Z", "mode": "train", "global_step": 676, "epoch": 0.02715186568662891, "loss": -0.1218, "grad_norm": 9.489598274230957, "learning_rate": 7.954545454545455e-06, "num_tokens": 1468025.0, "completions/mean_length": 45.5, "completions/min_length": 41.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9916675090789795, "rewards/meter/std": 0.0027789692394435406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9916675090789795, "rewards/total_composite/std": 0.0027789692394435406, "reward": 0.9916675090789795, "reward_std": 0.002778968308120966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040273357182741165, "sampling/sampling_logp_difference/max": 2.040062665939331, "sampling/importance_sampling_ratio/min": 0.13002057373523712, "sampling/importance_sampling_ratio/mean": 0.9964287877082825, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16463111247867346, "clip_ratio/low_mean": 0.009073751280084252, "clip_ratio/low_min": 0.009073751280084252, "clip_ratio/high_mean": 0.01704174862243235, "clip_ratio/high_max": 0.01704174862243235, "clip_ratio/region_mean": 0.026115499902516603, "reward_total_mean": 0.9916675090789795, "reward_meter_mean": 0.9916675090789795, "reward_meter_std": 0.0027789692394435406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9916675090789795, "reward_total_composite_std": 0.0027789692394435406, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1509.0} {"timestamp_utc": "2026-04-11T22:40:17Z", "mode": "train", "global_step": 677, "epoch": 0.027192031168413865, "loss": 0.0185, "grad_norm": 4.125744342803955, "learning_rate": 7.951515151515152e-06, "num_tokens": 1469690.0, "completions/mean_length": 62.125, "completions/min_length": 58.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9938265085220337, "rewards/meter/std": 0.0003937912406399846, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6625509858131409, "rewards/total_composite/std": 0.00026252749375998974, "reward": 0.6625509858131409, "reward_std": 0.0002625406195875257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02882186695933342, "sampling/sampling_logp_difference/max": 5.798013687133789, "sampling/importance_sampling_ratio/min": 0.0030335744377225637, "sampling/importance_sampling_ratio/mean": 1.0004545450210571, "sampling/importance_sampling_ratio/max": 1.5561105012893677, "entropy": 0.11069364938884974, "clip_ratio/low_mean": 0.012098872568458319, "clip_ratio/low_min": 0.012098872568458319, "clip_ratio/high_mean": 0.0021551724057644606, "clip_ratio/high_max": 0.0021551724057644606, "clip_ratio/region_mean": 0.01425404497422278, "reward_total_mean": 0.6625509858131409, "reward_meter_mean": 0.9938265085220337, "reward_meter_std": 0.0003937912406399846, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6625509858131409, "reward_total_composite_std": 0.00026252749375998974, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1510.0} {"timestamp_utc": "2026-04-11T22:40:27Z", "mode": "train", "global_step": 678, "epoch": 0.02723219665019882, "loss": -0.15, "grad_norm": 0.7712501287460327, "learning_rate": 7.948484848484848e-06, "num_tokens": 1472788.0, "completions/mean_length": 336.25, "completions/min_length": 268.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 277.66668701171875, "completions/min_terminated_length": 268.0, "completions/max_terminated_length": 291.0, "rewards/meter/mean": 0.6280694007873535, "rewards/meter/std": 0.5047115683555603, "rewards/count_adherence/mean": 0.8035714626312256, "rewards/count_adherence/std": 0.15152288973331451, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.41633522510528564, "rewards/repeat_penalty/std": 0.3482770323753357, "rewards/total_composite/mean": 0.1397983729839325, "rewards/total_composite/std": 0.1796947717666626, "reward": 0.1397983729839325, "reward_std": 0.1796947568655014, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01377029623836279, "sampling/sampling_logp_difference/max": 5.731827259063721, "sampling/importance_sampling_ratio/min": 0.003241149475798011, "sampling/importance_sampling_ratio/mean": 0.9978663921356201, "sampling/importance_sampling_ratio/max": 1.6066981554031372, "entropy": 0.03139376197941601, "clip_ratio/low_mean": 0.004044867469929159, "clip_ratio/low_min": 0.004044867469929159, "clip_ratio/high_mean": 0.002251501166028902, "clip_ratio/high_max": 0.002251501166028902, "clip_ratio/region_mean": 0.006296368635958061, "reward_total_mean": 0.1397983729839325, "reward_meter_mean": 0.6280694007873535, "reward_meter_std": 0.5047115683555603, "reward_count_adherence_mean": 0.8035714626312256, "reward_count_adherence_std": 0.15152288973331451, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.41633522510528564, "reward_repeat_penalty_std": 0.3482770323753357, "reward_total_composite_mean": 0.1397983729839325, "reward_total_composite_std": 0.1796947717666626, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1511.0} {"timestamp_utc": "2026-04-11T22:40:31Z", "mode": "train", "global_step": 679, "epoch": 0.027272362131983773, "loss": -0.0123, "grad_norm": 12.517946243286133, "learning_rate": 7.945454545454547e-06, "num_tokens": 1474604.0, "completions/mean_length": 57.0, "completions/min_length": 53.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9841916561126709, "rewards/meter/std": 0.005828971043229103, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9428657293319702, "rewards/total_composite/std": 0.1139112040400505, "reward": 0.9428657293319702, "reward_std": 0.1139112189412117, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02025281824171543, "sampling/sampling_logp_difference/max": 1.1112747192382812, "sampling/importance_sampling_ratio/min": 0.3291391432285309, "sampling/importance_sampling_ratio/mean": 0.9998968243598938, "sampling/importance_sampling_ratio/max": 1.880238652229309, "entropy": 0.11628487333655357, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/high_mean": 0.01512786210514605, "clip_ratio/high_max": 0.01512786210514605, "clip_ratio/region_mean": 0.017486352706328034, "reward_total_mean": 0.9428657293319702, "reward_meter_mean": 0.9841916561126709, "reward_meter_std": 0.005828971043229103, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9428657293319702, "reward_total_composite_std": 0.1139112040400505, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1512.0} {"timestamp_utc": "2026-04-11T22:40:36Z", "mode": "train", "global_step": 680, "epoch": 0.027312527613768726, "loss": 0.031, "grad_norm": 8.33311653137207, "learning_rate": 7.942424242424242e-06, "num_tokens": 1476777.0, "completions/mean_length": 115.625, "completions/min_length": 101.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.625, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.371232271194458, "rewards/meter/std": 0.47603076696395874, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.48125001788139343, "rewards/repeat_penalty/std": 0.13611315190792084, "rewards/total_composite/mean": 0.22119268774986267, "rewards/total_composite/std": 0.28687217831611633, "reward": 0.22119268774986267, "reward_std": 0.2868722081184387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020960384979844093, "sampling/sampling_logp_difference/max": 1.9988516569137573, "sampling/importance_sampling_ratio/min": 0.1354907900094986, "sampling/importance_sampling_ratio/mean": 1.0001816749572754, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08438334474340081, "clip_ratio/low_mean": 0.014545571291819215, "clip_ratio/low_min": 0.014545571291819215, "clip_ratio/high_mean": 0.0033385155256837606, "clip_ratio/high_max": 0.0033385155256837606, "clip_ratio/region_mean": 0.017884086817502975, "reward_total_mean": 0.22119268774986267, "reward_meter_mean": 0.371232271194458, "reward_meter_std": 0.47603076696395874, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.48125001788139343, "reward_repeat_penalty_std": 0.13611315190792084, "reward_total_composite_mean": 0.22119268774986267, "reward_total_composite_std": 0.28687217831611633, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1513.0} {"timestamp_utc": "2026-04-11T22:40:41Z", "mode": "train", "global_step": 681, "epoch": 0.02735269309555368, "loss": 0.0231, "grad_norm": 3.9222095012664795, "learning_rate": 7.93939393939394e-06, "num_tokens": 1478560.0, "completions/mean_length": 57.875, "completions/min_length": 57.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9723160266876221, "rewards/meter/std": 0.027899622917175293, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.8934845924377441, "rewards/total_composite/std": 0.1633896678686142, "reward": 0.8934845924377441, "reward_std": 0.163389652967453, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023344971239566803, "sampling/sampling_logp_difference/max": 1.0728743076324463, "sampling/importance_sampling_ratio/min": 0.342024028301239, "sampling/importance_sampling_ratio/mean": 1.0034892559051514, "sampling/importance_sampling_ratio/max": 1.6929512023925781, "entropy": 0.11835484858602285, "clip_ratio/low_mean": 0.014689265750348568, "clip_ratio/low_min": 0.014689265750348568, "clip_ratio/high_mean": 0.015127861872315407, "clip_ratio/high_max": 0.015127861872315407, "clip_ratio/region_mean": 0.029817127622663975, "reward_total_mean": 0.8934845924377441, "reward_meter_mean": 0.9723160266876221, "reward_meter_std": 0.027899622917175293, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.8934845924377441, "reward_total_composite_std": 0.1633896678686142, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1514.0} {"timestamp_utc": "2026-04-11T22:40:50Z", "mode": "train", "global_step": 682, "epoch": 0.027392858577338634, "loss": -0.0676, "grad_norm": 1.2039419412612915, "learning_rate": 7.936363636363637e-06, "num_tokens": 1480663.0, "completions/mean_length": 173.875, "completions/min_length": 109.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 125.5714340209961, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.38672682642936707, "rewards/meter/std": 0.36102262139320374, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.2519763112068176, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.48750001192092896, "rewards/repeat_penalty/std": 0.24604006111621857, "rewards/total_composite/mean": 0.16035160422325134, "rewards/total_composite/std": 0.24009783565998077, "reward": 0.16035160422325134, "reward_std": 0.24009782075881958, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013726767152547836, "sampling/sampling_logp_difference/max": 1.2122516632080078, "sampling/importance_sampling_ratio/min": 0.29752659797668457, "sampling/importance_sampling_ratio/mean": 1.0000630617141724, "sampling/importance_sampling_ratio/max": 1.6329307556152344, "entropy": 0.049124513287097216, "clip_ratio/low_mean": 0.005251961061730981, "clip_ratio/low_min": 0.005251961061730981, "clip_ratio/high_mean": 0.0029069767333567142, "clip_ratio/high_max": 0.0029069767333567142, "clip_ratio/region_mean": 0.008158937795087695, "reward_total_mean": 0.16035160422325134, "reward_meter_mean": 0.38672682642936707, "reward_meter_std": 0.36102262139320374, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.2519763112068176, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.48750001192092896, "reward_repeat_penalty_std": 0.24604006111621857, "reward_total_composite_mean": 0.16035160422325134, "reward_total_composite_std": 0.24009783565998077, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1515.0} {"timestamp_utc": "2026-04-11T22:40:55Z", "mode": "train", "global_step": 683, "epoch": 0.02743302405912359, "loss": 0.0096, "grad_norm": 4.130276679992676, "learning_rate": 7.933333333333334e-06, "num_tokens": 1482349.0, "completions/mean_length": 58.75, "completions/min_length": 57.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9239417314529419, "rewards/meter/std": 0.05428864806890488, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.8474001884460449, "rewards/total_composite/std": 0.15459680557250977, "reward": 0.8474001884460449, "reward_std": 0.15459680557250977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03419341892004013, "sampling/sampling_logp_difference/max": 1.8840758800506592, "sampling/importance_sampling_ratio/min": 0.15196943283081055, "sampling/importance_sampling_ratio/mean": 0.994130551815033, "sampling/importance_sampling_ratio/max": 1.6103789806365967, "entropy": 0.11382754053920507, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/high_mean": 0.016894312808290124, "clip_ratio/high_max": 0.016894312808290124, "clip_ratio/region_mean": 0.02113160095177591, "reward_total_mean": 0.8474001884460449, "reward_meter_mean": 0.9239417314529419, "reward_meter_std": 0.05428864806890488, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.8474001884460449, "reward_total_composite_std": 0.15459680557250977, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1516.0} {"timestamp_utc": "2026-04-11T22:40:59Z", "mode": "train", "global_step": 684, "epoch": 0.027473189540908542, "loss": -0.0064, "grad_norm": 4.445869445800781, "learning_rate": 7.930303030303031e-06, "num_tokens": 1484327.0, "completions/mean_length": 80.25, "completions/min_length": 79.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.25, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9331251978874207, "rewards/meter/std": 0.07388782501220703, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.10690450668334961, "rewards/total_composite/mean": 0.46393126249313354, "rewards/total_composite/std": 0.09499123692512512, "reward": 0.46393126249313354, "reward_std": 0.09499124437570572, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01861550658941269, "sampling/sampling_logp_difference/max": 1.1865754127502441, "sampling/importance_sampling_ratio/min": 0.3052648901939392, "sampling/importance_sampling_ratio/mean": 1.0021919012069702, "sampling/importance_sampling_ratio/max": 1.7777718305587769, "entropy": 0.0625843945890665, "clip_ratio/low_mean": 0.007758883642964065, "clip_ratio/low_min": 0.007758883642964065, "clip_ratio/high_mean": 0.004629629664123058, "clip_ratio/high_max": 0.004629629664123058, "clip_ratio/region_mean": 0.012388513307087123, "reward_total_mean": 0.46393126249313354, "reward_meter_mean": 0.9331251978874207, "reward_meter_std": 0.07388782501220703, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.10690450668334961, "reward_total_composite_mean": 0.46393126249313354, "reward_total_composite_std": 0.09499123692512512, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1517.0} {"timestamp_utc": "2026-04-11T22:41:05Z", "mode": "train", "global_step": 685, "epoch": 0.027513355022693496, "loss": 0.0202, "grad_norm": 2.300926685333252, "learning_rate": 7.927272727272729e-06, "num_tokens": 1487252.0, "completions/mean_length": 170.625, "completions/min_length": 165.0, "completions/max_length": 178.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.625, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 178.0, "rewards/meter/mean": 0.8037871718406677, "rewards/meter/std": 0.32751503586769104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4107142686843872, "rewards/repeat_penalty/std": 0.050507623702287674, "rewards/total_composite/mean": 0.34419798851013184, "rewards/total_composite/std": 0.14113976061344147, "reward": 0.34419798851013184, "reward_std": 0.14113974571228027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010442078113555908, "sampling/sampling_logp_difference/max": 1.643384337425232, "sampling/importance_sampling_ratio/min": 0.19332465529441833, "sampling/importance_sampling_ratio/mean": 0.9983497262001038, "sampling/importance_sampling_ratio/max": 1.4589576721191406, "entropy": 0.046010758727788925, "clip_ratio/low_mean": 0.0007022471982054412, "clip_ratio/low_min": 0.0007022471982054412, "clip_ratio/high_mean": 0.010335821425542235, "clip_ratio/high_max": 0.010335821425542235, "clip_ratio/region_mean": 0.011038068623747677, "reward_total_mean": 0.34419798851013184, "reward_meter_mean": 0.8037871718406677, "reward_meter_std": 0.32751503586769104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4107142686843872, "reward_repeat_penalty_std": 0.050507623702287674, "reward_total_composite_mean": 0.34419798851013184, "reward_total_composite_std": 0.14113976061344147, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1518.0} {"timestamp_utc": "2026-04-11T22:41:15Z", "mode": "train", "global_step": 686, "epoch": 0.02755352050447845, "loss": -0.0998, "grad_norm": 1.047049641609192, "learning_rate": 7.924242424242426e-06, "num_tokens": 1490154.0, "completions/mean_length": 251.75, "completions/min_length": 212.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 214.57144165039062, "completions/min_terminated_length": 212.0, "completions/max_terminated_length": 222.0, "rewards/meter/mean": 0.9798320531845093, "rewards/meter/std": 0.047762468457221985, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.2828427255153656, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.32499998807907104, "rewards/repeat_penalty/std": 0.27645719051361084, "rewards/total_composite/mean": 0.2209167182445526, "rewards/total_composite/std": 0.04932519793510437, "reward": 0.2209167182445526, "reward_std": 0.04932519420981407, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011242986656725407, "sampling/sampling_logp_difference/max": 2.929636001586914, "sampling/importance_sampling_ratio/min": 0.053416479378938675, "sampling/importance_sampling_ratio/mean": 0.9986944198608398, "sampling/importance_sampling_ratio/max": 1.7827941179275513, "entropy": 0.021744283847510815, "clip_ratio/low_mean": 0.003458057180978358, "clip_ratio/low_min": 0.003458057180978358, "clip_ratio/high_mean": 0.00234195904340595, "clip_ratio/high_max": 0.00234195904340595, "clip_ratio/region_mean": 0.005800016224384308, "reward_total_mean": 0.2209167182445526, "reward_meter_mean": 0.9798320531845093, "reward_meter_std": 0.047762468457221985, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.2828427255153656, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.32499998807907104, "reward_repeat_penalty_std": 0.27645719051361084, "reward_total_composite_mean": 0.2209167182445526, "reward_total_composite_std": 0.04932519793510437, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1519.0} {"timestamp_utc": "2026-04-11T22:41:19Z", "mode": "train", "global_step": 687, "epoch": 0.027593685986263404, "loss": 0.0292, "grad_norm": 10.655113220214844, "learning_rate": 7.921212121212122e-06, "num_tokens": 1491874.0, "completions/mean_length": 71.0, "completions/min_length": 67.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8845878839492798, "rewards/meter/std": 0.2909509837627411, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.8430095314979553, "rewards/total_composite/std": 0.2961682975292206, "reward": 0.8430095314979553, "reward_std": 0.2961682677268982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05931292846798897, "sampling/sampling_logp_difference/max": 2.937185287475586, "sampling/importance_sampling_ratio/min": 0.05301474407315254, "sampling/importance_sampling_ratio/mean": 1.0036027431488037, "sampling/importance_sampling_ratio/max": 1.9307304620742798, "entropy": 0.1976525131613016, "clip_ratio/low_mean": 0.011955027701333165, "clip_ratio/low_min": 0.011955027701333165, "clip_ratio/high_mean": 0.020970338257029653, "clip_ratio/high_max": 0.020970338257029653, "clip_ratio/region_mean": 0.03292536595836282, "reward_total_mean": 0.8430095314979553, "reward_meter_mean": 0.8845878839492798, "reward_meter_std": 0.2909509837627411, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.8430095314979553, "reward_total_composite_std": 0.2961682975292206, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1520.0} {"timestamp_utc": "2026-04-11T22:41:24Z", "mode": "train", "global_step": 688, "epoch": 0.027633851468048358, "loss": -0.0086, "grad_norm": 1.4717203378677368, "learning_rate": 7.918181818181819e-06, "num_tokens": 1493736.0, "completions/mean_length": 67.75, "completions/min_length": 66.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9984875321388245, "rewards/meter/std": 0.0001689638738753274, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6656583547592163, "rewards/total_composite/std": 0.00011263116175541654, "reward": 0.6656583547592163, "reward_std": 0.00011264239583397284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018908310681581497, "sampling/sampling_logp_difference/max": 2.604221820831299, "sampling/importance_sampling_ratio/min": 0.07396066933870316, "sampling/importance_sampling_ratio/mean": 0.9985640048980713, "sampling/importance_sampling_ratio/max": 1.6862038373947144, "entropy": 0.06369170360267162, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/high_mean": 0.010716472752392292, "clip_ratio/high_max": 0.010716472752392292, "clip_ratio/region_mean": 0.02208010945469141, "reward_total_mean": 0.6656583547592163, "reward_meter_mean": 0.9984875321388245, "reward_meter_std": 0.0001689638738753274, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6656583547592163, "reward_total_composite_std": 0.00011263116175541654, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1521.0} {"timestamp_utc": "2026-04-11T22:41:30Z", "mode": "train", "global_step": 689, "epoch": 0.027674016949833312, "loss": 0.0309, "grad_norm": 2.0693576335906982, "learning_rate": 7.915151515151516e-06, "num_tokens": 1497021.0, "completions/mean_length": 202.625, "completions/min_length": 191.0, "completions/max_length": 212.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 202.625, "completions/min_terminated_length": 191.0, "completions/max_terminated_length": 212.0, "rewards/meter/mean": 0.995306134223938, "rewards/meter/std": 0.0053658634424209595, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4194444417953491, "rewards/repeat_penalty/std": 0.13975918292999268, "rewards/total_composite/mean": 0.4050329029560089, "rewards/total_composite/std": 0.1355670541524887, "reward": 0.4050329029560089, "reward_std": 0.1355670541524887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014271133579313755, "sampling/sampling_logp_difference/max": 2.4142937660217285, "sampling/importance_sampling_ratio/min": 0.08943047374486923, "sampling/importance_sampling_ratio/mean": 0.9986175298690796, "sampling/importance_sampling_ratio/max": 1.674434781074524, "entropy": 0.03671248443424702, "clip_ratio/low_mean": 0.007414351915940642, "clip_ratio/low_min": 0.007414351915940642, "clip_ratio/high_mean": 0.0019430051324889064, "clip_ratio/high_max": 0.0019430051324889064, "clip_ratio/region_mean": 0.009357357048429549, "reward_total_mean": 0.4050329029560089, "reward_meter_mean": 0.995306134223938, "reward_meter_std": 0.0053658634424209595, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4194444417953491, "reward_repeat_penalty_std": 0.13975918292999268, "reward_total_composite_mean": 0.4050329029560089, "reward_total_composite_std": 0.1355670541524887, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1522.0} {"timestamp_utc": "2026-04-11T22:41:39Z", "mode": "train", "global_step": 690, "epoch": 0.027714182431618266, "loss": 0.0024, "grad_norm": 1.1219055652618408, "learning_rate": 7.912121212121213e-06, "num_tokens": 1502074.0, "completions/mean_length": 401.625, "completions/min_length": 400.0, "completions/max_length": 402.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 401.625, "completions/min_terminated_length": 400.0, "completions/max_terminated_length": 402.0, "rewards/meter/mean": 0.9981971979141235, "rewards/meter/std": 0.000627980858553201, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.2503289580345154, "rewards/repeat_penalty/std": 0.23873279988765717, "rewards/total_composite/mean": 0.1784874051809311, "rewards/total_composite/std": 0.17026527225971222, "reward": 0.1784874051809311, "reward_std": 0.17026525735855103, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005141077097505331, "sampling/sampling_logp_difference/max": 1.3938775062561035, "sampling/importance_sampling_ratio/min": 0.24811138212680817, "sampling/importance_sampling_ratio/mean": 1.0005240440368652, "sampling/importance_sampling_ratio/max": 1.6704438924789429, "entropy": 0.027160495053976774, "clip_ratio/low_mean": 0.0021797263179905713, "clip_ratio/low_min": 0.0021797263179905713, "clip_ratio/high_mean": 0.0015578281017951667, "clip_ratio/high_max": 0.0015578281017951667, "clip_ratio/region_mean": 0.003737554419785738, "reward_total_mean": 0.1784874051809311, "reward_meter_mean": 0.9981971979141235, "reward_meter_std": 0.000627980858553201, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.2503289580345154, "reward_repeat_penalty_std": 0.23873279988765717, "reward_total_composite_mean": 0.1784874051809311, "reward_total_composite_std": 0.17026527225971222, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1523.0} {"timestamp_utc": "2026-04-11T22:41:43Z", "mode": "train", "global_step": 691, "epoch": 0.02775434791340322, "loss": -0.1031, "grad_norm": 11.69098949432373, "learning_rate": 7.909090909090909e-06, "num_tokens": 1503707.0, "completions/mean_length": 38.125, "completions/min_length": 19.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.125, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.4370739161968231, "rewards/meter/std": 0.2700711488723755, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4370739161968231, "rewards/total_composite/std": 0.2700711488723755, "reward": 0.4370739161968231, "reward_std": 0.2700711488723755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0835486352443695, "sampling/sampling_logp_difference/max": 1.102844476699829, "sampling/importance_sampling_ratio/min": 0.3319256007671356, "sampling/importance_sampling_ratio/mean": 1.0089696645736694, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5944820679724216, "clip_ratio/low_mean": 0.019354344811290503, "clip_ratio/low_min": 0.019354344811290503, "clip_ratio/high_mean": 0.04530784301459789, "clip_ratio/high_max": 0.04530784301459789, "clip_ratio/region_mean": 0.0646621878258884, "reward_total_mean": 0.4370739161968231, "reward_meter_mean": 0.4370739161968231, "reward_meter_std": 0.2700711488723755, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4370739161968231, "reward_total_composite_std": 0.2700711488723755, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1524.0} {"timestamp_utc": "2026-04-11T22:41:47Z", "mode": "train", "global_step": 692, "epoch": 0.027794513395188174, "loss": 0.0153, "grad_norm": 5.031876564025879, "learning_rate": 7.906060606060608e-06, "num_tokens": 1505712.0, "completions/mean_length": 86.625, "completions/min_length": 84.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.625, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9942313432693481, "rewards/meter/std": 0.0012215422466397285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942313432693481, "rewards/total_composite/std": 0.0012215422466397285, "reward": 0.9942313432693481, "reward_std": 0.00122154806740582, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005889675114303827, "sampling/sampling_logp_difference/max": 0.44769424200057983, "sampling/importance_sampling_ratio/min": 0.7906866669654846, "sampling/importance_sampling_ratio/mean": 1.003043293952942, "sampling/importance_sampling_ratio/max": 1.5647001266479492, "entropy": 0.04205932654440403, "clip_ratio/low_mean": 0.0014367816038429737, "clip_ratio/low_min": 0.0014367816038429737, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0014367816038429737, "reward_total_mean": 0.9942313432693481, "reward_meter_mean": 0.9942313432693481, "reward_meter_std": 0.0012215422466397285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9942313432693481, "reward_total_composite_std": 0.0012215422466397285, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1525.0} {"timestamp_utc": "2026-04-11T22:41:52Z", "mode": "train", "global_step": 693, "epoch": 0.027834678876973128, "loss": 0.0041, "grad_norm": 10.364151954650879, "learning_rate": 7.903030303030303e-06, "num_tokens": 1507385.0, "completions/mean_length": 54.125, "completions/min_length": 52.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9447178244590759, "rewards/meter/std": 0.019572317600250244, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9447178244590759, "rewards/total_composite/std": 0.019572317600250244, "reward": 0.9447178244590759, "reward_std": 0.01957232505083084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024356268346309662, "sampling/sampling_logp_difference/max": 1.9013309478759766, "sampling/importance_sampling_ratio/min": 0.14936968684196472, "sampling/importance_sampling_ratio/mean": 1.003846287727356, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0895584006793797, "clip_ratio/low_mean": 0.009437322150915861, "clip_ratio/low_min": 0.009437322150915861, "clip_ratio/high_mean": 0.01416083937510848, "clip_ratio/high_max": 0.01416083937510848, "clip_ratio/region_mean": 0.02359816152602434, "reward_total_mean": 0.9447178244590759, "reward_meter_mean": 0.9447178244590759, "reward_meter_std": 0.019572317600250244, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9447178244590759, "reward_total_composite_std": 0.019572317600250244, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1526.0} {"timestamp_utc": "2026-04-11T22:41:56Z", "mode": "train", "global_step": 694, "epoch": 0.02787484435875808, "loss": 0.0035, "grad_norm": 6.670380115509033, "learning_rate": 7.9e-06, "num_tokens": 1509273.0, "completions/mean_length": 74.0, "completions/min_length": 70.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.610145092010498, "rewards/meter/std": 0.23973627388477325, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.5444035530090332, "rewards/total_composite/std": 0.24921227991580963, "reward": 0.5444035530090332, "reward_std": 0.24921227991580963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030294643715023994, "sampling/sampling_logp_difference/max": 1.8091429471969604, "sampling/importance_sampling_ratio/min": 0.16379445791244507, "sampling/importance_sampling_ratio/mean": 1.0047415494918823, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17548873648047447, "clip_ratio/low_mean": 0.01502489356789738, "clip_ratio/low_min": 0.01502489356789738, "clip_ratio/high_mean": 0.02390445303171873, "clip_ratio/high_max": 0.02390445303171873, "clip_ratio/region_mean": 0.03892934659961611, "reward_total_mean": 0.5444035530090332, "reward_meter_mean": 0.610145092010498, "reward_meter_std": 0.23973627388477325, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.5444035530090332, "reward_total_composite_std": 0.24921227991580963, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1527.0} {"timestamp_utc": "2026-04-11T22:42:01Z", "mode": "train", "global_step": 695, "epoch": 0.02791500984054304, "loss": -0.0124, "grad_norm": 3.7013967037200928, "learning_rate": 7.896969696969698e-06, "num_tokens": 1511199.0, "completions/mean_length": 80.75, "completions/min_length": 78.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.75, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9964468479156494, "rewards/meter/std": 0.0019429969834163785, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.788793683052063, "rewards/total_composite/std": 0.17156179249286652, "reward": 0.788793683052063, "reward_std": 0.17156179249286652, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012971381656825542, "sampling/sampling_logp_difference/max": 1.7286226749420166, "sampling/importance_sampling_ratio/min": 0.1775287538766861, "sampling/importance_sampling_ratio/mean": 1.0024663209915161, "sampling/importance_sampling_ratio/max": 1.8497363328933716, "entropy": 0.06276643788442016, "clip_ratio/low_mean": 0.0062915480230003595, "clip_ratio/low_min": 0.0062915480230003595, "clip_ratio/high_mean": 0.004360056365840137, "clip_ratio/high_max": 0.004360056365840137, "clip_ratio/region_mean": 0.010651604388840497, "reward_total_mean": 0.788793683052063, "reward_meter_mean": 0.9964468479156494, "reward_meter_std": 0.0019429969834163785, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.788793683052063, "reward_total_composite_std": 0.17156179249286652, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1528.0} {"timestamp_utc": "2026-04-11T22:42:06Z", "mode": "train", "global_step": 696, "epoch": 0.027955175322327993, "loss": -0.0063, "grad_norm": 3.119380235671997, "learning_rate": 7.893939393939395e-06, "num_tokens": 1513573.0, "completions/mean_length": 125.75, "completions/min_length": 112.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.75, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9228010177612305, "rewards/meter/std": 0.20124337077140808, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7000000476837158, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.6532999277114868, "rewards/total_composite/std": 0.18960076570510864, "reward": 0.6532999277114868, "reward_std": 0.18960076570510864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016659488901495934, "sampling/sampling_logp_difference/max": 1.3884687423706055, "sampling/importance_sampling_ratio/min": 0.2494570016860962, "sampling/importance_sampling_ratio/mean": 0.9984103441238403, "sampling/importance_sampling_ratio/max": 1.9363112449645996, "entropy": 0.07676348416134715, "clip_ratio/low_mean": 0.0020850637229159474, "clip_ratio/low_min": 0.0020850637229159474, "clip_ratio/high_mean": 0.00876707280986011, "clip_ratio/high_max": 0.00876707280986011, "clip_ratio/region_mean": 0.010852136532776058, "reward_total_mean": 0.6532999277114868, "reward_meter_mean": 0.9228010177612305, "reward_meter_std": 0.20124337077140808, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7000000476837158, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.6532999277114868, "reward_total_composite_std": 0.18960076570510864, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1529.0} {"timestamp_utc": "2026-04-11T22:42:11Z", "mode": "train", "global_step": 697, "epoch": 0.027995340804112947, "loss": 0.0606, "grad_norm": 4.277256011962891, "learning_rate": 7.89090909090909e-06, "num_tokens": 1515360.0, "completions/mean_length": 59.375, "completions/min_length": 55.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.375, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9654116630554199, "rewards/meter/std": 0.034304678440093994, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.8894380331039429, "rewards/total_composite/std": 0.17405925691127777, "reward": 0.8894380331039429, "reward_std": 0.17405925691127777, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020010532811284065, "sampling/sampling_logp_difference/max": 1.0107874870300293, "sampling/importance_sampling_ratio/min": 0.3639322817325592, "sampling/importance_sampling_ratio/mean": 0.99860018491745, "sampling/importance_sampling_ratio/max": 1.4003076553344727, "entropy": 0.09744885191321373, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/high_mean": 0.01933896285481751, "clip_ratio/high_max": 0.01933896285481751, "clip_ratio/region_mean": 0.023185116704553366, "reward_total_mean": 0.8894380331039429, "reward_meter_mean": 0.9654116630554199, "reward_meter_std": 0.034304678440093994, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.8894380331039429, "reward_total_composite_std": 0.17405925691127777, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1530.0} {"timestamp_utc": "2026-04-11T22:42:21Z", "mode": "train", "global_step": 698, "epoch": 0.0280355062858979, "loss": -0.0873, "grad_norm": 2.2967092990875244, "learning_rate": 7.88787878787879e-06, "num_tokens": 1519450.0, "completions/mean_length": 382.25, "completions/min_length": 321.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 363.71429443359375, "completions/min_terminated_length": 321.0, "completions/max_terminated_length": 412.0, "rewards/meter/mean": 0.6105729937553406, "rewards/meter/std": 0.4790833294391632, "rewards/count_adherence/mean": 0.8854166865348816, "rewards/count_adherence/std": 0.1254950612783432, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.3964124917984009, "rewards/repeat_penalty/std": 0.25513792037963867, "rewards/total_composite/mean": 0.18387337028980255, "rewards/total_composite/std": 0.21844197809696198, "reward": 0.18387337028980255, "reward_std": 0.21844197809696198, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0396236851811409, "sampling/sampling_logp_difference/max": 6.693481922149658, "sampling/importance_sampling_ratio/min": 0.001238961354829371, "sampling/importance_sampling_ratio/mean": 0.9987672567367554, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18515709601342678, "clip_ratio/low_mean": 0.007161757908761501, "clip_ratio/low_min": 0.007161757908761501, "clip_ratio/high_mean": 0.015907755587249994, "clip_ratio/high_max": 0.015907755587249994, "clip_ratio/region_mean": 0.023069513496011496, "reward_total_mean": 0.18387337028980255, "reward_meter_mean": 0.6105729937553406, "reward_meter_std": 0.4790833294391632, "reward_count_adherence_mean": 0.8854166865348816, "reward_count_adherence_std": 0.1254950612783432, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.3964124917984009, "reward_repeat_penalty_std": 0.25513792037963867, "reward_total_composite_mean": 0.18387337028980255, "reward_total_composite_std": 0.21844197809696198, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1531.0} {"timestamp_utc": "2026-04-11T22:42:25Z", "mode": "train", "global_step": 699, "epoch": 0.028075671767682855, "loss": 0.0056, "grad_norm": 8.473615646362305, "learning_rate": 7.884848484848485e-06, "num_tokens": 1521331.0, "completions/mean_length": 67.125, "completions/min_length": 63.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9927859306335449, "rewards/meter/std": 0.014541360549628735, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.8282062411308289, "rewards/total_composite/std": 0.18183693289756775, "reward": 0.8282062411308289, "reward_std": 0.18183691799640656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03374454379081726, "sampling/sampling_logp_difference/max": 1.766160249710083, "sampling/importance_sampling_ratio/min": 0.1795266717672348, "sampling/importance_sampling_ratio/mean": 0.9997567534446716, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14588068891316652, "clip_ratio/low_mean": 0.00747219193726778, "clip_ratio/low_min": 0.00747219193726778, "clip_ratio/high_mean": 0.02036348171532154, "clip_ratio/high_max": 0.02036348171532154, "clip_ratio/region_mean": 0.02783567365258932, "reward_total_mean": 0.8282062411308289, "reward_meter_mean": 0.9927859306335449, "reward_meter_std": 0.014541360549628735, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.8282062411308289, "reward_total_composite_std": 0.18183693289756775, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1532.0} {"timestamp_utc": "2026-04-11T22:42:35Z", "mode": "train", "global_step": 700, "epoch": 0.02811583724946781, "loss": 0.0097, "grad_norm": 0.8298696279525757, "learning_rate": 7.881818181818182e-06, "num_tokens": 1526242.0, "completions/mean_length": 414.875, "completions/min_length": 401.0, "completions/max_length": 444.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 414.875, "completions/min_terminated_length": 401.0, "completions/max_terminated_length": 444.0, "rewards/meter/mean": 0.9960967302322388, "rewards/meter/std": 0.0013218529056757689, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.029462797567248344, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.19121241569519043, "rewards/repeat_penalty/std": 0.07494427263736725, "rewards/total_composite/mean": 0.16019713878631592, "rewards/total_composite/std": 0.061176449060440063, "reward": 0.16019713878631592, "reward_std": 0.061176449060440063, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008506695739924908, "sampling/sampling_logp_difference/max": 1.432920217514038, "sampling/importance_sampling_ratio/min": 0.238611102104187, "sampling/importance_sampling_ratio/mean": 1.0012154579162598, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.045434954110533, "clip_ratio/low_mean": 0.00364298140630126, "clip_ratio/low_min": 0.00364298140630126, "clip_ratio/high_mean": 0.006157157360576093, "clip_ratio/high_max": 0.006157157360576093, "clip_ratio/region_mean": 0.009800138766877353, "reward_total_mean": 0.16019713878631592, "reward_meter_mean": 0.9960967302322388, "reward_meter_std": 0.0013218529056757689, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.029462797567248344, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.19121241569519043, "reward_repeat_penalty_std": 0.07494427263736725, "reward_total_composite_mean": 0.16019713878631592, "reward_total_composite_std": 0.061176449060440063, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1533.0} {"timestamp_utc": "2026-04-11T22:44:02Z", "mode": "eval", "global_step": 700, "epoch": 0.02811583724946781, "eval_loss": NaN, "eval_runtime": 87.7418, "eval_samples_per_second": 1.185, "eval_steps_per_second": 0.148, "eval_num_tokens": 1526242.0, "eval_completions/mean_length": 249.43269230769232, "eval_completions/min_length": 68.53846153846153, "eval_completions/max_length": 471.9230769230769, "eval_completions/clipped_ratio": 0.11538461538461539, "eval_completions/mean_terminated_length": 217.6739994929387, "eval_completions/min_terminated_length": 68.53846153846153, "eval_completions/max_terminated_length": 415.15384615384613, "eval_rewards/meter/mean": 0.6321157022164419, "eval_rewards/meter/std": 0.37711624113413006, "eval_rewards/count_adherence/mean": 0.8881359283740704, "eval_rewards/count_adherence/std": 0.16248861929545036, "eval_rewards/arabic_clean/mean": 0.9038461538461539, "eval_rewards/arabic_clean/std": 0.2343954168833219, "eval_rewards/repeat_penalty/mean": 0.5953293947073129, "eval_rewards/repeat_penalty/std": 0.32899803152451146, "eval_rewards/total_composite/mean": 0.29438196466519284, "eval_rewards/total_composite/std": 0.28008361991781455, "eval_reward": 0.29438196466519284, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.012457216982371531, "eval_sampling/sampling_logp_difference/max": 0.881988103573139, "eval_sampling/importance_sampling_ratio/min": 0.4343368663237645, "eval_sampling/importance_sampling_ratio/mean": 1.0041690973135142, "eval_sampling/importance_sampling_ratio/max": 1.455959943624643, "eval_entropy": 0.1376804428604933, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.29438196466519284, "eval_reward_meter_mean": 0.6321157022164419, "eval_reward_meter_std": 0.37711624113413006, "eval_reward_count_adherence_mean": 0.8881359283740704, "eval_reward_count_adherence_std": 0.16248861929545036, "eval_reward_arabic_clean_mean": 0.9038461538461539, "eval_reward_arabic_clean_std": 0.2343954168833219, "eval_reward_repeat_penalty_mean": 0.5953293947073129, "eval_reward_repeat_penalty_std": 0.32899803152451146, "eval_reward_total_composite_mean": 0.29438196466519284, "eval_reward_total_composite_std": 0.28008361991781455, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1533.0} {"timestamp_utc": "2026-04-11T22:44:17Z", "mode": "train", "global_step": 701, "epoch": 0.028156002731252763, "loss": -0.12, "grad_norm": 2.5312631130218506, "learning_rate": 7.87878787878788e-06, "num_tokens": 1528168.0, "completions/mean_length": 138.75, "completions/min_length": 79.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 85.42857360839844, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.646858811378479, "rewards/meter/std": 0.40135496854782104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6387161016464233, "rewards/total_composite/std": 0.41526278853416443, "reward": 0.6387161016464233, "reward_std": 0.41526278853416443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017966095358133316, "sampling/sampling_logp_difference/max": 0.638648509979248, "sampling/importance_sampling_ratio/min": 0.5280055403709412, "sampling/importance_sampling_ratio/mean": 1.0036530494689941, "sampling/importance_sampling_ratio/max": 1.6849335432052612, "entropy": 0.11331802047789097, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.015958538744598627, "clip_ratio/high_max": 0.015958538744598627, "clip_ratio/region_mean": 0.015958538744598627, "reward_total_mean": 0.6387161016464233, "reward_meter_mean": 0.646858811378479, "reward_meter_std": 0.40135496854782104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6387161016464233, "reward_total_composite_std": 0.41526278853416443, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1534.0} {"timestamp_utc": "2026-04-11T22:44:27Z", "mode": "train", "global_step": 702, "epoch": 0.028196168213037717, "loss": -0.0691, "grad_norm": 3.0773770809173584, "learning_rate": 7.875757575757577e-06, "num_tokens": 1529770.0, "completions/mean_length": 128.25, "completions/min_length": 71.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.42857360839844, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.3803873062133789, "rewards/meter/std": 0.26946189999580383, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.33759811520576477, "rewards/total_composite/std": 0.30163025856018066, "reward": 0.33759811520576477, "reward_std": 0.3016302287578583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04749147966504097, "sampling/sampling_logp_difference/max": 1.9366450309753418, "sampling/importance_sampling_ratio/min": 0.14418688416481018, "sampling/importance_sampling_ratio/mean": 1.011560082435608, "sampling/importance_sampling_ratio/max": 1.7729606628417969, "entropy": 0.37723080068826675, "clip_ratio/low_mean": 0.01317842910066247, "clip_ratio/low_min": 0.01317842910066247, "clip_ratio/high_mean": 0.0155344782397151, "clip_ratio/high_max": 0.0155344782397151, "clip_ratio/region_mean": 0.02871290734037757, "reward_total_mean": 0.33759811520576477, "reward_meter_mean": 0.3803873062133789, "reward_meter_std": 0.26946189999580383, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.33759811520576477, "reward_total_composite_std": 0.30163025856018066, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1535.0} {"timestamp_utc": "2026-04-11T22:44:34Z", "mode": "train", "global_step": 703, "epoch": 0.02823633369482267, "loss": -0.0245, "grad_norm": 1.2964930534362793, "learning_rate": 7.872727272727273e-06, "num_tokens": 1533320.0, "completions/mean_length": 229.75, "completions/min_length": 206.0, "completions/max_length": 242.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 229.75, "completions/min_terminated_length": 206.0, "completions/max_terminated_length": 242.0, "rewards/meter/mean": 0.9970081448554993, "rewards/meter/std": 0.001187506248243153, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.2613636255264282, "rewards/repeat_penalty/std": 0.197011336684227, "rewards/total_composite/mean": 0.2607119679450989, "rewards/total_composite/std": 0.19680176675319672, "reward": 0.2607119679450989, "reward_std": 0.1968017816543579, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014087834395468235, "sampling/sampling_logp_difference/max": 4.6370849609375, "sampling/importance_sampling_ratio/min": 0.009685891680419445, "sampling/importance_sampling_ratio/mean": 0.9988393187522888, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03166929807048291, "clip_ratio/low_mean": 0.0028346364269964397, "clip_ratio/low_min": 0.0028346364269964397, "clip_ratio/high_mean": 0.004737977171316743, "clip_ratio/high_max": 0.004737977171316743, "clip_ratio/region_mean": 0.007572613598313183, "reward_total_mean": 0.2607119679450989, "reward_meter_mean": 0.9970081448554993, "reward_meter_std": 0.001187506248243153, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.2613636255264282, "reward_repeat_penalty_std": 0.197011336684227, "reward_total_composite_mean": 0.2607119679450989, "reward_total_composite_std": 0.19680176675319672, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1536.0} {"timestamp_utc": "2026-04-11T22:44:44Z", "mode": "train", "global_step": 704, "epoch": 0.028276499176607624, "loss": -0.1074, "grad_norm": 3.015087127685547, "learning_rate": 7.86969696969697e-06, "num_tokens": 1536668.0, "completions/mean_length": 364.5, "completions/min_length": 229.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 315.3333435058594, "completions/min_terminated_length": 229.0, "completions/max_terminated_length": 368.0, "rewards/meter/mean": 0.20065514743328094, "rewards/meter/std": 0.3053584098815918, "rewards/count_adherence/mean": 0.8522727489471436, "rewards/count_adherence/std": 0.21697448194026947, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.6299689412117004, "rewards/repeat_penalty/std": 0.28235092759132385, "rewards/total_composite/mean": 0.0969100296497345, "rewards/total_composite/std": 0.21232596039772034, "reward": 0.0969100296497345, "reward_std": 0.21232594549655914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04316805675625801, "sampling/sampling_logp_difference/max": 2.4648847579956055, "sampling/importance_sampling_ratio/min": 0.08501863479614258, "sampling/importance_sampling_ratio/mean": 0.9969695806503296, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2017015889286995, "clip_ratio/low_mean": 0.01385743310675025, "clip_ratio/low_min": 0.01385743310675025, "clip_ratio/high_mean": 0.009915342554450035, "clip_ratio/high_max": 0.009915342554450035, "clip_ratio/region_mean": 0.023772775661200285, "reward_total_mean": 0.0969100296497345, "reward_meter_mean": 0.20065514743328094, "reward_meter_std": 0.3053584098815918, "reward_count_adherence_mean": 0.8522727489471436, "reward_count_adherence_std": 0.21697448194026947, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.6299689412117004, "reward_repeat_penalty_std": 0.28235092759132385, "reward_total_composite_mean": 0.0969100296497345, "reward_total_composite_std": 0.21232596039772034, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1537.0} {"timestamp_utc": "2026-04-11T22:44:49Z", "mode": "train", "global_step": 705, "epoch": 0.02831666465839258, "loss": -0.0075, "grad_norm": 2.615363836288452, "learning_rate": 7.866666666666667e-06, "num_tokens": 1538402.0, "completions/mean_length": 58.75, "completions/min_length": 56.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9921605587005615, "rewards/meter/std": 0.006017809733748436, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.909184455871582, "rewards/total_composite/std": 0.15155236423015594, "reward": 0.909184455871582, "reward_std": 0.15155236423015594, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021948276087641716, "sampling/sampling_logp_difference/max": 3.973776340484619, "sampling/importance_sampling_ratio/min": 0.01880229450762272, "sampling/importance_sampling_ratio/mean": 0.9994084239006042, "sampling/importance_sampling_ratio/max": 1.5749088525772095, "entropy": 0.053199955029413104, "clip_ratio/low_mean": 0.006398809840902686, "clip_ratio/low_min": 0.006398809840902686, "clip_ratio/high_mean": 0.006355932215228677, "clip_ratio/high_max": 0.006355932215228677, "clip_ratio/region_mean": 0.012754742056131363, "reward_total_mean": 0.909184455871582, "reward_meter_mean": 0.9921605587005615, "reward_meter_std": 0.006017809733748436, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.909184455871582, "reward_total_composite_std": 0.15155236423015594, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1538.0} {"timestamp_utc": "2026-04-11T22:44:53Z", "mode": "train", "global_step": 706, "epoch": 0.028356830140177532, "loss": 0.0093, "grad_norm": 4.515374183654785, "learning_rate": 7.863636363636364e-06, "num_tokens": 1539816.0, "completions/mean_length": 44.75, "completions/min_length": 43.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9157871603965759, "rewards/meter/std": 0.05085242539644241, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9157871603965759, "rewards/total_composite/std": 0.05085242539644241, "reward": 0.9157871603965759, "reward_std": 0.05085243284702301, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02820507250726223, "sampling/sampling_logp_difference/max": 0.7927889823913574, "sampling/importance_sampling_ratio/min": 0.4525808095932007, "sampling/importance_sampling_ratio/mean": 0.997020423412323, "sampling/importance_sampling_ratio/max": 1.2550123929977417, "entropy": 0.13661748263984919, "clip_ratio/low_mean": 0.0027777778450399637, "clip_ratio/low_min": 0.0027777778450399637, "clip_ratio/high_mean": 0.01414728700183332, "clip_ratio/high_max": 0.01414728700183332, "clip_ratio/region_mean": 0.016925064846873283, "reward_total_mean": 0.9157871603965759, "reward_meter_mean": 0.9157871603965759, "reward_meter_std": 0.05085242539644241, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9157871603965759, "reward_total_composite_std": 0.05085242539644241, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1539.0} {"timestamp_utc": "2026-04-11T22:44:59Z", "mode": "train", "global_step": 707, "epoch": 0.028396995621962486, "loss": -0.0003, "grad_norm": 2.2196364402770996, "learning_rate": 7.860606060606062e-06, "num_tokens": 1541901.0, "completions/mean_length": 84.625, "completions/min_length": 83.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.625, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.9956607818603516, "rewards/meter/std": 0.0007934165187180042, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956607818603516, "rewards/total_composite/std": 0.0007934165187180042, "reward": 0.9956607818603516, "reward_std": 0.0007934237364679575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026080820709466934, "sampling/sampling_logp_difference/max": 0.8360247611999512, "sampling/importance_sampling_ratio/min": 0.4566366374492645, "sampling/importance_sampling_ratio/mean": 1.0091147422790527, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15222040470689535, "clip_ratio/low_mean": 0.007405462441965938, "clip_ratio/low_min": 0.007405462441965938, "clip_ratio/high_mean": 0.013377037481404841, "clip_ratio/high_max": 0.013377037481404841, "clip_ratio/region_mean": 0.02078249992337078, "reward_total_mean": 0.9956607818603516, "reward_meter_mean": 0.9956607818603516, "reward_meter_std": 0.0007934165187180042, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9956607818603516, "reward_total_composite_std": 0.0007934165187180042, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1540.0} {"timestamp_utc": "2026-04-11T22:45:07Z", "mode": "train", "global_step": 708, "epoch": 0.02843716110374744, "loss": 0.0303, "grad_norm": 2.581402063369751, "learning_rate": 7.857575757575759e-06, "num_tokens": 1546099.0, "completions/mean_length": 330.75, "completions/min_length": 284.0, "completions/max_length": 358.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 330.75, "completions/min_terminated_length": 284.0, "completions/max_terminated_length": 358.0, "rewards/meter/mean": 0.5674466490745544, "rewards/meter/std": 0.4153376519680023, "rewards/count_adherence/mean": 0.9027777910232544, "rewards/count_adherence/std": 0.03928370773792267, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.34049707651138306, "rewards/repeat_penalty/std": 0.22640277445316315, "rewards/total_composite/mean": 0.2035118043422699, "rewards/total_composite/std": 0.23429380357265472, "reward": 0.2035118043422699, "reward_std": 0.23429378867149353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023940538987517357, "sampling/sampling_logp_difference/max": 3.625192880630493, "sampling/importance_sampling_ratio/min": 0.026643957942724228, "sampling/importance_sampling_ratio/mean": 1.002025842666626, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10461874585598707, "clip_ratio/low_mean": 0.012675938894972205, "clip_ratio/low_min": 0.012675938894972205, "clip_ratio/high_mean": 0.010587403550744057, "clip_ratio/high_max": 0.010587403550744057, "clip_ratio/region_mean": 0.023263342445716262, "reward_total_mean": 0.2035118043422699, "reward_meter_mean": 0.5674466490745544, "reward_meter_std": 0.4153376519680023, "reward_count_adherence_mean": 0.9027777910232544, "reward_count_adherence_std": 0.03928370773792267, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.34049707651138306, "reward_repeat_penalty_std": 0.22640277445316315, "reward_total_composite_mean": 0.2035118043422699, "reward_total_composite_std": 0.23429380357265472, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1541.0} {"timestamp_utc": "2026-04-11T22:45:17Z", "mode": "train", "global_step": 709, "epoch": 0.028477326585532394, "loss": -0.3582, "grad_norm": 0.8571996688842773, "learning_rate": 7.854545454545454e-06, "num_tokens": 1549805.0, "completions/mean_length": 452.25, "completions/min_length": 372.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 416.3999938964844, "completions/min_terminated_length": 372.0, "completions/max_terminated_length": 472.0, "rewards/meter/mean": 0.8675932884216309, "rewards/meter/std": 0.35074320435523987, "rewards/count_adherence/mean": 0.6590909361839294, "rewards/count_adherence/std": 0.39101481437683105, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/repeat_penalty/mean": 0.6495236158370972, "rewards/repeat_penalty/std": 0.30929210782051086, "rewards/total_composite/mean": 0.25732704997062683, "rewards/total_composite/std": 0.2531158924102783, "reward": 0.25732704997062683, "reward_std": 0.2531158924102783, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019949674606323242, "sampling/sampling_logp_difference/max": 2.8196310997009277, "sampling/importance_sampling_ratio/min": 0.05962793529033661, "sampling/importance_sampling_ratio/mean": 1.0035980939865112, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0702343238517642, "clip_ratio/low_mean": 0.0015743073308840394, "clip_ratio/low_min": 0.0015743073308840394, "clip_ratio/high_mean": 0.005005840037483722, "clip_ratio/high_max": 0.005005840037483722, "clip_ratio/region_mean": 0.006580147368367761, "reward_total_mean": 0.25732704997062683, "reward_meter_mean": 0.8675932884216309, "reward_meter_std": 0.35074320435523987, "reward_count_adherence_mean": 0.6590909361839294, "reward_count_adherence_std": 0.39101481437683105, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_repeat_penalty_mean": 0.6495236158370972, "reward_repeat_penalty_std": 0.30929210782051086, "reward_total_composite_mean": 0.25732704997062683, "reward_total_composite_std": 0.2531158924102783, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1542.0} {"timestamp_utc": "2026-04-11T22:45:28Z", "mode": "train", "global_step": 710, "epoch": 0.028517492067317348, "loss": -0.2247, "grad_norm": 0.6216875314712524, "learning_rate": 7.851515151515152e-06, "num_tokens": 1554988.0, "completions/mean_length": 474.875, "completions/min_length": 439.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 469.5714416503906, "completions/min_terminated_length": 439.0, "completions/max_terminated_length": 497.0, "rewards/meter/mean": 0.9897267818450928, "rewards/meter/std": 0.019744135439395905, "rewards/count_adherence/mean": 0.7767857313156128, "rewards/count_adherence/std": 0.20360276103019714, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5224603414535522, "rewards/repeat_penalty/std": 0.23697951436042786, "rewards/total_composite/mean": 0.3333708643913269, "rewards/total_composite/std": 0.18023891746997833, "reward": 0.3333708643913269, "reward_std": 0.18023891746997833, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010584630072116852, "sampling/sampling_logp_difference/max": 2.4926586151123047, "sampling/importance_sampling_ratio/min": 0.08268983662128448, "sampling/importance_sampling_ratio/mean": 1.0017552375793457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05670151812955737, "clip_ratio/low_mean": 0.0021170872496441007, "clip_ratio/low_min": 0.0021170872496441007, "clip_ratio/high_mean": 0.004514876694884151, "clip_ratio/high_max": 0.004514876694884151, "clip_ratio/region_mean": 0.006631963944528252, "reward_total_mean": 0.3333708643913269, "reward_meter_mean": 0.9897267818450928, "reward_meter_std": 0.019744135439395905, "reward_count_adherence_mean": 0.7767857313156128, "reward_count_adherence_std": 0.20360276103019714, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5224603414535522, "reward_repeat_penalty_std": 0.23697951436042786, "reward_total_composite_mean": 0.3333708643913269, "reward_total_composite_std": 0.18023891746997833, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1543.0} {"timestamp_utc": "2026-04-11T22:45:38Z", "mode": "train", "global_step": 711, "epoch": 0.028557657549102302, "loss": -0.0049, "grad_norm": 0.6046677231788635, "learning_rate": 7.848484848484849e-06, "num_tokens": 1560550.0, "completions/mean_length": 440.25, "completions/min_length": 436.0, "completions/max_length": 461.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 440.25, "completions/min_terminated_length": 436.0, "completions/max_terminated_length": 461.0, "rewards/meter/mean": 0.9979845285415649, "rewards/meter/std": 0.00018585202633403242, "rewards/count_adherence/mean": 0.7946428060531616, "rewards/count_adherence/std": 0.025253823027014732, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.0676877498626709, "rewards/repeat_penalty/std": 0.023803479969501495, "rewards/total_composite/mean": 0.053851306438446045, "rewards/total_composite/std": 0.019493911415338516, "reward": 0.053851306438446045, "reward_std": 0.019493909552693367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003577793249860406, "sampling/sampling_logp_difference/max": 1.1886392831802368, "sampling/importance_sampling_ratio/min": 0.30463549494743347, "sampling/importance_sampling_ratio/mean": 1.0001322031021118, "sampling/importance_sampling_ratio/max": 1.460545539855957, "entropy": 0.011412000167183578, "clip_ratio/low_mean": 0.0008561643480788916, "clip_ratio/low_min": 0.0008561643480788916, "clip_ratio/high_mean": 0.0016890882980078459, "clip_ratio/high_max": 0.0016890882980078459, "clip_ratio/region_mean": 0.0025452526460867375, "reward_total_mean": 0.053851306438446045, "reward_meter_mean": 0.9979845285415649, "reward_meter_std": 0.00018585202633403242, "reward_count_adherence_mean": 0.7946428060531616, "reward_count_adherence_std": 0.025253823027014732, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.0676877498626709, "reward_repeat_penalty_std": 0.023803479969501495, "reward_total_composite_mean": 0.053851306438446045, "reward_total_composite_std": 0.019493911415338516, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1544.0} {"timestamp_utc": "2026-04-11T22:45:44Z", "mode": "train", "global_step": 712, "epoch": 0.028597823030887256, "loss": 0.0004, "grad_norm": 4.805282115936279, "learning_rate": 7.845454545454546e-06, "num_tokens": 1562410.0, "completions/mean_length": 69.5, "completions/min_length": 68.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.8851493000984192, "rewards/meter/std": 0.1646534502506256, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7055873274803162, "rewards/total_composite/std": 0.21653573215007782, "reward": 0.7055873274803162, "reward_std": 0.216535747051239, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03851144388318062, "sampling/sampling_logp_difference/max": 1.9700713157653809, "sampling/importance_sampling_ratio/min": 0.1394468992948532, "sampling/importance_sampling_ratio/mean": 0.997905969619751, "sampling/importance_sampling_ratio/max": 1.8839155435562134, "entropy": 0.1954718241468072, "clip_ratio/low_mean": 0.016306631732732058, "clip_ratio/low_min": 0.016306631732732058, "clip_ratio/high_mean": 0.014235412469133735, "clip_ratio/high_max": 0.014235412469133735, "clip_ratio/region_mean": 0.030542044201865792, "reward_total_mean": 0.7055873274803162, "reward_meter_mean": 0.8851493000984192, "reward_meter_std": 0.1646534502506256, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.7055873274803162, "reward_total_composite_std": 0.21653573215007782, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1545.0} {"timestamp_utc": "2026-04-11T22:45:54Z", "mode": "train", "global_step": 713, "epoch": 0.02863798851267221, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.842424242424243e-06, "num_tokens": 1564234.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9704335927963257, "rewards/meter/std": 0.07359233498573303, "rewards/count_adherence/mean": 0.8166667222976685, "rewards/count_adherence/std": 0.030860668048262596, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5789903998374939, "rewards/repeat_penalty/std": 0.045423366129398346, "rewards/total_composite/mean": 0.4566306471824646, "rewards/total_composite/std": 0.02297402359545231, "reward": 0.4566306471824646, "reward_std": 0.02297401800751686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.4566306471824646, "reward_meter_mean": 0.9704335927963257, "reward_meter_std": 0.07359233498573303, "reward_count_adherence_mean": 0.8166667222976685, "reward_count_adherence_std": 0.030860668048262596, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5789903998374939, "reward_repeat_penalty_std": 0.045423366129398346, "reward_total_composite_mean": 0.4566306471824646, "reward_total_composite_std": 0.02297402359545231, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1546.0} {"timestamp_utc": "2026-04-11T22:46:02Z", "mode": "train", "global_step": 714, "epoch": 0.028678153994457164, "loss": 0.0251, "grad_norm": 0.7193350791931152, "learning_rate": 7.83939393939394e-06, "num_tokens": 1568398.0, "completions/mean_length": 336.5, "completions/min_length": 311.0, "completions/max_length": 345.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 336.5, "completions/min_terminated_length": 311.0, "completions/max_terminated_length": 345.0, "rewards/meter/mean": 0.9977849721908569, "rewards/meter/std": 0.0006831432110629976, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.19117647409439087, "rewards/repeat_penalty/std": 0.16262187063694, "rewards/total_composite/mean": 0.13619501888751984, "rewards/total_composite/std": 0.1156935766339302, "reward": 0.13619501888751984, "reward_std": 0.11569356918334961, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003576630027964711, "sampling/sampling_logp_difference/max": 1.290938377380371, "sampling/importance_sampling_ratio/min": 0.2750125825405121, "sampling/importance_sampling_ratio/mean": 0.9997386932373047, "sampling/importance_sampling_ratio/max": 1.3743659257888794, "entropy": 0.018028545891866088, "clip_ratio/low_mean": 0.0018522579048294574, "clip_ratio/low_min": 0.0018522579048294574, "clip_ratio/high_mean": 0.0012019231216982007, "clip_ratio/high_max": 0.0012019231216982007, "clip_ratio/region_mean": 0.003054181026527658, "reward_total_mean": 0.13619501888751984, "reward_meter_mean": 0.9977849721908569, "reward_meter_std": 0.0006831432110629976, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.19117647409439087, "reward_repeat_penalty_std": 0.16262187063694, "reward_total_composite_mean": 0.13619501888751984, "reward_total_composite_std": 0.1156935766339302, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1547.0} {"timestamp_utc": "2026-04-11T22:46:08Z", "mode": "train", "global_step": 715, "epoch": 0.028718319476242118, "loss": 0.0035, "grad_norm": 4.87160062789917, "learning_rate": 7.836363636363638e-06, "num_tokens": 1570787.0, "completions/mean_length": 123.625, "completions/min_length": 115.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.625, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9901050329208374, "rewards/meter/std": 0.0037112515419721603, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.1608559489250183, "rewards/total_composite/mean": 0.7246866822242737, "rewards/total_composite/std": 0.15871340036392212, "reward": 0.7246866822242737, "reward_std": 0.15871338546276093, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04571465030312538, "sampling/sampling_logp_difference/max": 1.6292939186096191, "sampling/importance_sampling_ratio/min": 0.19606797397136688, "sampling/importance_sampling_ratio/mean": 0.9998616576194763, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23239293694496155, "clip_ratio/low_mean": 0.023657660058233887, "clip_ratio/low_min": 0.023657660058233887, "clip_ratio/high_mean": 0.018382353708148003, "clip_ratio/high_max": 0.018382353708148003, "clip_ratio/region_mean": 0.04204001376638189, "reward_total_mean": 0.7246866822242737, "reward_meter_mean": 0.9901050329208374, "reward_meter_std": 0.0037112515419721603, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.1608559489250183, "reward_total_composite_mean": 0.7246866822242737, "reward_total_composite_std": 0.15871340036392212, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1548.0} {"timestamp_utc": "2026-04-11T22:46:14Z", "mode": "train", "global_step": 716, "epoch": 0.02875848495802707, "loss": 0.0175, "grad_norm": 2.7540860176086426, "learning_rate": 7.833333333333333e-06, "num_tokens": 1573577.0, "completions/mean_length": 175.75, "completions/min_length": 162.0, "completions/max_length": 215.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 175.75, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 215.0, "rewards/meter/mean": 0.8235251903533936, "rewards/meter/std": 0.28830382227897644, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6091269850730896, "rewards/repeat_penalty/std": 0.19641855359077454, "rewards/total_composite/mean": 0.4638897478580475, "rewards/total_composite/std": 0.2275657206773758, "reward": 0.4638897478580475, "reward_std": 0.2275657057762146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02091868966817856, "sampling/sampling_logp_difference/max": 2.359158992767334, "sampling/importance_sampling_ratio/min": 0.09449966251850128, "sampling/importance_sampling_ratio/mean": 1.0048706531524658, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15365072712302208, "clip_ratio/low_mean": 0.010836752247996628, "clip_ratio/low_min": 0.010836752247996628, "clip_ratio/high_mean": 0.0044064579415135086, "clip_ratio/high_max": 0.0044064579415135086, "clip_ratio/region_mean": 0.015243210189510137, "reward_total_mean": 0.4638897478580475, "reward_meter_mean": 0.8235251903533936, "reward_meter_std": 0.28830382227897644, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6091269850730896, "reward_repeat_penalty_std": 0.19641855359077454, "reward_total_composite_mean": 0.4638897478580475, "reward_total_composite_std": 0.2275657206773758, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1549.0} {"timestamp_utc": "2026-04-11T22:46:24Z", "mode": "train", "global_step": 717, "epoch": 0.028798650439812026, "loss": -0.1657, "grad_norm": 1.3285834789276123, "learning_rate": 7.83030303030303e-06, "num_tokens": 1575801.0, "completions/mean_length": 391.0, "completions/min_length": 156.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 189.33334350585938, "completions/min_terminated_length": 156.0, "completions/max_terminated_length": 210.0, "rewards/meter/mean": 0.3074490427970886, "rewards/meter/std": 0.2025083303451538, "rewards/count_adherence/mean": 0.824999988079071, "rewards/count_adherence/std": 0.12817399203777313, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/repeat_penalty/mean": 0.8395833373069763, "rewards/repeat_penalty/std": 0.25571832060813904, "rewards/total_composite/mean": 0.08441510796546936, "rewards/total_composite/std": 0.10946666449308395, "reward": 0.08441510796546936, "reward_std": 0.10946667194366455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05243195965886116, "sampling/sampling_logp_difference/max": 8.712602615356445, "sampling/importance_sampling_ratio/min": 0.0001644995791139081, "sampling/importance_sampling_ratio/mean": 1.002199649810791, "sampling/importance_sampling_ratio/max": 1.6081585884094238, "entropy": 0.13471947237849236, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.010442280676215887, "clip_ratio/high_max": 0.010442280676215887, "clip_ratio/region_mean": 0.010442280676215887, "reward_total_mean": 0.08441510796546936, "reward_meter_mean": 0.3074490427970886, "reward_meter_std": 0.2025083303451538, "reward_count_adherence_mean": 0.824999988079071, "reward_count_adherence_std": 0.12817399203777313, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_repeat_penalty_mean": 0.8395833373069763, "reward_repeat_penalty_std": 0.25571832060813904, "reward_total_composite_mean": 0.08441510796546936, "reward_total_composite_std": 0.10946666449308395, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1550.0} {"timestamp_utc": "2026-04-11T22:46:29Z", "mode": "train", "global_step": 718, "epoch": 0.02883881592159698, "loss": 0.0058, "grad_norm": 6.637593746185303, "learning_rate": 7.827272727272728e-06, "num_tokens": 1577519.0, "completions/mean_length": 57.75, "completions/min_length": 57.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9619162082672119, "rewards/meter/std": 0.06612562388181686, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9619162082672119, "rewards/total_composite/std": 0.06612562388181686, "reward": 0.9619162082672119, "reward_std": 0.06612562388181686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03136257454752922, "sampling/sampling_logp_difference/max": 1.5830154418945312, "sampling/importance_sampling_ratio/min": 0.20535492897033691, "sampling/importance_sampling_ratio/mean": 1.004167914390564, "sampling/importance_sampling_ratio/max": 1.769487977027893, "entropy": 0.14472659677267075, "clip_ratio/low_mean": 0.006465517450124025, "clip_ratio/low_min": 0.006465517450124025, "clip_ratio/high_mean": 0.023710741428658366, "clip_ratio/high_max": 0.023710741428658366, "clip_ratio/region_mean": 0.03017625887878239, "reward_total_mean": 0.9619162082672119, "reward_meter_mean": 0.9619162082672119, "reward_meter_std": 0.06612562388181686, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9619162082672119, "reward_total_composite_std": 0.06612562388181686, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1551.0} {"timestamp_utc": "2026-04-11T22:46:39Z", "mode": "train", "global_step": 719, "epoch": 0.028878981403381934, "loss": -0.1972, "grad_norm": 0.982711136341095, "learning_rate": 7.824242424242425e-06, "num_tokens": 1580014.0, "completions/mean_length": 208.875, "completions/min_length": 142.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 165.57144165039062, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.8615807294845581, "rewards/meter/std": 0.1756223738193512, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5022321939468384, "rewards/repeat_penalty/std": 0.19195686280727386, "rewards/total_composite/mean": 0.36472243070602417, "rewards/total_composite/std": 0.1904788464307785, "reward": 0.36472243070602417, "reward_std": 0.1904788315296173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011406843550503254, "sampling/sampling_logp_difference/max": 0.5394062995910645, "sampling/importance_sampling_ratio/min": 0.5830943584442139, "sampling/importance_sampling_ratio/mean": 1.0052262544631958, "sampling/importance_sampling_ratio/max": 1.6336287260055542, "entropy": 0.07858177460730076, "clip_ratio/low_mean": 0.0008802816737443209, "clip_ratio/low_min": 0.0008802816737443209, "clip_ratio/high_mean": 0.005198180675506592, "clip_ratio/high_max": 0.005198180675506592, "clip_ratio/region_mean": 0.006078462349250913, "reward_total_mean": 0.36472243070602417, "reward_meter_mean": 0.8615807294845581, "reward_meter_std": 0.1756223738193512, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5022321939468384, "reward_repeat_penalty_std": 0.19195686280727386, "reward_total_composite_mean": 0.36472243070602417, "reward_total_composite_std": 0.1904788464307785, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1552.0} {"timestamp_utc": "2026-04-11T22:46:47Z", "mode": "train", "global_step": 720, "epoch": 0.028919146885166887, "loss": -0.0478, "grad_norm": 1.5212147235870361, "learning_rate": 7.821212121212122e-06, "num_tokens": 1583419.0, "completions/mean_length": 228.625, "completions/min_length": 195.0, "completions/max_length": 243.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 228.625, "completions/min_terminated_length": 195.0, "completions/max_terminated_length": 243.0, "rewards/meter/mean": 0.9813181161880493, "rewards/meter/std": 0.009512215852737427, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.06613000482320786, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.527634859085083, "rewards/repeat_penalty/std": 0.18670302629470825, "rewards/total_composite/mean": 0.3818710744380951, "rewards/total_composite/std": 0.13756081461906433, "reward": 0.3818710744380951, "reward_std": 0.13756079971790314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014808963052928448, "sampling/sampling_logp_difference/max": 1.9443225860595703, "sampling/importance_sampling_ratio/min": 0.14308412373065948, "sampling/importance_sampling_ratio/mean": 1.0003083944320679, "sampling/importance_sampling_ratio/max": 1.7837820053100586, "entropy": 0.08757321583107114, "clip_ratio/low_mean": 0.0012820513220503926, "clip_ratio/low_min": 0.0012820513220503926, "clip_ratio/high_mean": 0.015587894711643457, "clip_ratio/high_max": 0.015587894711643457, "clip_ratio/region_mean": 0.01686994603369385, "reward_total_mean": 0.3818710744380951, "reward_meter_mean": 0.9813181161880493, "reward_meter_std": 0.009512215852737427, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.06613000482320786, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.527634859085083, "reward_repeat_penalty_std": 0.18670302629470825, "reward_total_composite_mean": 0.3818710744380951, "reward_total_composite_std": 0.13756081461906433, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1553.0} {"timestamp_utc": "2026-04-11T22:46:52Z", "mode": "train", "global_step": 721, "epoch": 0.02895931236695184, "loss": -0.0114, "grad_norm": 9.25263786315918, "learning_rate": 7.81818181818182e-06, "num_tokens": 1585267.0, "completions/mean_length": 65.0, "completions/min_length": 60.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.6475868821144104, "rewards/meter/std": 0.3421926498413086, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.5946333408355713, "rewards/total_composite/std": 0.34638410806655884, "reward": 0.5946333408355713, "reward_std": 0.34638410806655884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1203828752040863, "sampling/sampling_logp_difference/max": 2.941412925720215, "sampling/importance_sampling_ratio/min": 0.052791085094213486, "sampling/importance_sampling_ratio/mean": 0.991131603717804, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5916384495794773, "clip_ratio/low_mean": 0.051155281253159046, "clip_ratio/low_min": 0.051155281253159046, "clip_ratio/high_mean": 0.05381742771714926, "clip_ratio/high_max": 0.05381742771714926, "clip_ratio/region_mean": 0.1049727089703083, "reward_total_mean": 0.5946333408355713, "reward_meter_mean": 0.6475868821144104, "reward_meter_std": 0.3421926498413086, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.5946333408355713, "reward_total_composite_std": 0.34638410806655884, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1554.0} {"timestamp_utc": "2026-04-11T22:46:57Z", "mode": "train", "global_step": 722, "epoch": 0.028999477848736795, "loss": 0.3273, "grad_norm": 7.886468410491943, "learning_rate": 7.815151515151515e-06, "num_tokens": 1586837.0, "completions/mean_length": 48.25, "completions/min_length": 42.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.9944906830787659, "rewards/meter/std": 0.0011895333882421255, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8698122501373291, "rewards/total_composite/std": 0.35145723819732666, "reward": 0.8698122501373291, "reward_std": 0.35145723819732666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01547156646847725, "sampling/sampling_logp_difference/max": 0.7411696910858154, "sampling/importance_sampling_ratio/min": 0.4765561819076538, "sampling/importance_sampling_ratio/mean": 1.0014286041259766, "sampling/importance_sampling_ratio/max": 1.3288551568984985, "entropy": 0.10689277853816748, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8698122501373291, "reward_meter_mean": 0.9944906830787659, "reward_meter_std": 0.0011895333882421255, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8698122501373291, "reward_total_composite_std": 0.35145723819732666, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1555.0} {"timestamp_utc": "2026-04-11T22:47:08Z", "mode": "train", "global_step": 723, "epoch": 0.02903964333052175, "loss": -0.2014, "grad_norm": 0.8348656892776489, "learning_rate": 7.812121212121213e-06, "num_tokens": 1590899.0, "completions/mean_length": 485.75, "completions/min_length": 452.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 470.0, "completions/min_terminated_length": 452.0, "completions/max_terminated_length": 494.0, "rewards/meter/mean": 0.996362566947937, "rewards/meter/std": 0.0027744658291339874, "rewards/count_adherence/mean": 0.9464285373687744, "rewards/count_adherence/std": 0.03306501731276512, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.10470085591077805, "rewards/repeat_penalty/std": 0.10408195108175278, "rewards/total_composite/mean": 0.09887628257274628, "rewards/total_composite/std": 0.09698235988616943, "reward": 0.09887628257274628, "reward_std": 0.09698235988616943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007093369495123625, "sampling/sampling_logp_difference/max": 1.990865707397461, "sampling/importance_sampling_ratio/min": 0.13657712936401367, "sampling/importance_sampling_ratio/mean": 1.0003700256347656, "sampling/importance_sampling_ratio/max": 1.536428451538086, "entropy": 0.018016068963333964, "clip_ratio/low_mean": 0.0007961043156683445, "clip_ratio/low_min": 0.0007961043156683445, "clip_ratio/high_mean": 0.0007826215587556362, "clip_ratio/high_max": 0.0007826215587556362, "clip_ratio/region_mean": 0.0015787258744239807, "reward_total_mean": 0.09887628257274628, "reward_meter_mean": 0.996362566947937, "reward_meter_std": 0.0027744658291339874, "reward_count_adherence_mean": 0.9464285373687744, "reward_count_adherence_std": 0.03306501731276512, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.10470085591077805, "reward_repeat_penalty_std": 0.10408195108175278, "reward_total_composite_mean": 0.09887628257274628, "reward_total_composite_std": 0.09698235988616943, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1556.0} {"timestamp_utc": "2026-04-11T22:47:17Z", "mode": "train", "global_step": 724, "epoch": 0.029079808812306703, "loss": 0.0117, "grad_norm": 1.2350369691848755, "learning_rate": 7.80909090909091e-06, "num_tokens": 1595768.0, "completions/mean_length": 390.625, "completions/min_length": 379.0, "completions/max_length": 403.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 390.625, "completions/min_terminated_length": 379.0, "completions/max_terminated_length": 403.0, "rewards/meter/mean": 0.9978216886520386, "rewards/meter/std": 0.0007184247369877994, "rewards/count_adherence/mean": 0.734375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.17863407731056213, "rewards/repeat_penalty/std": 0.16910506784915924, "rewards/total_composite/mean": 0.1255086362361908, "rewards/total_composite/std": 0.10828039050102234, "reward": 0.1255086362361908, "reward_std": 0.10828038305044174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0057006035931408405, "sampling/sampling_logp_difference/max": 1.5134329795837402, "sampling/importance_sampling_ratio/min": 0.22015291452407837, "sampling/importance_sampling_ratio/mean": 0.9993754625320435, "sampling/importance_sampling_ratio/max": 1.8996793031692505, "entropy": 0.028950548847205937, "clip_ratio/low_mean": 0.0028556776233017445, "clip_ratio/low_min": 0.0028556776233017445, "clip_ratio/high_mean": 0.002920488826930523, "clip_ratio/high_max": 0.002920488826930523, "clip_ratio/region_mean": 0.005776166450232267, "reward_total_mean": 0.1255086362361908, "reward_meter_mean": 0.9978216886520386, "reward_meter_std": 0.0007184247369877994, "reward_count_adherence_mean": 0.734375, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.17863407731056213, "reward_repeat_penalty_std": 0.16910506784915924, "reward_total_composite_mean": 0.1255086362361908, "reward_total_composite_std": 0.10828039050102234, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1557.0} {"timestamp_utc": "2026-04-11T22:47:26Z", "mode": "train", "global_step": 725, "epoch": 0.029119974294091657, "loss": -0.1213, "grad_norm": 1.6505030393600464, "learning_rate": 7.806060606060607e-06, "num_tokens": 1597686.0, "completions/mean_length": 136.75, "completions/min_length": 78.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 83.14286041259766, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.5868990421295166, "rewards/meter/std": 0.38611555099487305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.5530991554260254, "rewards/total_composite/std": 0.40852445363998413, "reward": 0.5530991554260254, "reward_std": 0.4085244834423065, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022672384977340698, "sampling/sampling_logp_difference/max": 0.911320686340332, "sampling/importance_sampling_ratio/min": 0.4019929766654968, "sampling/importance_sampling_ratio/mean": 1.0019892454147339, "sampling/importance_sampling_ratio/max": 1.7016674280166626, "entropy": 0.09931796230375767, "clip_ratio/low_mean": 0.004787406767718494, "clip_ratio/low_min": 0.004787406767718494, "clip_ratio/high_mean": 0.007217321544885635, "clip_ratio/high_max": 0.007217321544885635, "clip_ratio/region_mean": 0.01200472831260413, "reward_total_mean": 0.5530991554260254, "reward_meter_mean": 0.5868990421295166, "reward_meter_std": 0.38611555099487305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.5530991554260254, "reward_total_composite_std": 0.40852445363998413, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1558.0} {"timestamp_utc": "2026-04-11T22:47:32Z", "mode": "train", "global_step": 726, "epoch": 0.02916013977587661, "loss": -0.0023, "grad_norm": 1.0429009199142456, "learning_rate": 7.803030303030303e-06, "num_tokens": 1600008.0, "completions/mean_length": 127.25, "completions/min_length": 126.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.25, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9302235841751099, "rewards/meter/std": 0.015423719771206379, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.1511857956647873, "rewards/total_composite/mean": 0.46501004695892334, "rewards/total_composite/std": 0.14159858226776123, "reward": 0.46501004695892334, "reward_std": 0.14159856736660004, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01078812312334776, "sampling/sampling_logp_difference/max": 0.9157018661499023, "sampling/importance_sampling_ratio/min": 0.4002356231212616, "sampling/importance_sampling_ratio/mean": 1.002754807472229, "sampling/importance_sampling_ratio/max": 1.5183614492416382, "entropy": 0.05762754753232002, "clip_ratio/low_mean": 0.0029605674790218472, "clip_ratio/low_min": 0.0029605674790218472, "clip_ratio/high_mean": 0.0009765625, "clip_ratio/high_max": 0.0009765625, "clip_ratio/region_mean": 0.003937129979021847, "reward_total_mean": 0.46501004695892334, "reward_meter_mean": 0.9302235841751099, "reward_meter_std": 0.015423719771206379, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.1511857956647873, "reward_total_composite_mean": 0.46501004695892334, "reward_total_composite_std": 0.14159858226776123, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1559.0} {"timestamp_utc": "2026-04-11T22:47:37Z", "mode": "train", "global_step": 727, "epoch": 0.029200305257661565, "loss": 0.0279, "grad_norm": 7.388904094696045, "learning_rate": 7.800000000000002e-06, "num_tokens": 1601923.0, "completions/mean_length": 99.375, "completions/min_length": 93.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.375, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.4179952144622803, "rewards/meter/std": 0.45823782682418823, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.3363860249519348, "rewards/total_composite/std": 0.36587125062942505, "reward": 0.3363860249519348, "reward_std": 0.36587125062942505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0847281664609909, "sampling/sampling_logp_difference/max": 5.028669357299805, "sampling/importance_sampling_ratio/min": 0.006547517143189907, "sampling/importance_sampling_ratio/mean": 0.9850558042526245, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2684608269482851, "clip_ratio/low_mean": 0.032587712397798896, "clip_ratio/low_min": 0.032587712397798896, "clip_ratio/high_mean": 0.02533797360956669, "clip_ratio/high_max": 0.02533797360956669, "clip_ratio/region_mean": 0.057925686007365584, "reward_total_mean": 0.3363860249519348, "reward_meter_mean": 0.4179952144622803, "reward_meter_std": 0.45823782682418823, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.3363860249519348, "reward_total_composite_std": 0.36587125062942505, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1560.0} {"timestamp_utc": "2026-04-11T22:47:46Z", "mode": "train", "global_step": 728, "epoch": 0.02924047073944652, "loss": 0.0076, "grad_norm": 2.0805506706237793, "learning_rate": 7.796969696969697e-06, "num_tokens": 1607332.0, "completions/mean_length": 459.125, "completions/min_length": 427.0, "completions/max_length": 483.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 459.125, "completions/min_terminated_length": 427.0, "completions/max_terminated_length": 483.0, "rewards/meter/mean": 0.28689688444137573, "rewards/meter/std": 0.3819184899330139, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0431290864944458, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5729086995124817, "rewards/repeat_penalty/std": 0.018385794013738632, "rewards/total_composite/mean": 0.12284211814403534, "rewards/total_composite/std": 0.1601785272359848, "reward": 0.12284211814403534, "reward_std": 0.1601785272359848, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019973881542682648, "sampling/sampling_logp_difference/max": 3.37805438041687, "sampling/importance_sampling_ratio/min": 0.03411376476287842, "sampling/importance_sampling_ratio/mean": 0.9993253350257874, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06644301256164908, "clip_ratio/low_mean": 0.008432107453700155, "clip_ratio/low_min": 0.008432107453700155, "clip_ratio/high_mean": 0.003258531214669347, "clip_ratio/high_max": 0.003258531214669347, "clip_ratio/region_mean": 0.011690638668369502, "reward_total_mean": 0.12284211814403534, "reward_meter_mean": 0.28689688444137573, "reward_meter_std": 0.3819184899330139, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0431290864944458, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5729086995124817, "reward_repeat_penalty_std": 0.018385794013738632, "reward_total_composite_mean": 0.12284211814403534, "reward_total_composite_std": 0.1601785272359848, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1561.0} {"timestamp_utc": "2026-04-11T22:47:56Z", "mode": "train", "global_step": 729, "epoch": 0.029280636221231473, "loss": -0.1369, "grad_norm": 2.659188747406006, "learning_rate": 7.793939393939394e-06, "num_tokens": 1609182.0, "completions/mean_length": 139.25, "completions/min_length": 80.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 86.00000762939453, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.700344443321228, "rewards/meter/std": 0.42837557196617126, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6998236179351807, "rewards/total_composite/std": 0.4293443560600281, "reward": 0.6998236179351807, "reward_std": 0.4293443262577057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022995855659246445, "sampling/sampling_logp_difference/max": 0.6324782371520996, "sampling/importance_sampling_ratio/min": 0.5312735438346863, "sampling/importance_sampling_ratio/mean": 1.0047773122787476, "sampling/importance_sampling_ratio/max": 1.5019832849502563, "entropy": 0.1982464650645852, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/high_mean": 0.007218867307528853, "clip_ratio/high_max": 0.007218867307528853, "clip_ratio/region_mean": 0.008781367330811918, "reward_total_mean": 0.6998236179351807, "reward_meter_mean": 0.700344443321228, "reward_meter_std": 0.42837557196617126, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6998236179351807, "reward_total_composite_std": 0.4293443560600281, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1562.0} {"timestamp_utc": "2026-04-11T22:48:01Z", "mode": "train", "global_step": 730, "epoch": 0.029320801703016427, "loss": 0.0222, "grad_norm": 3.446493625640869, "learning_rate": 7.790909090909092e-06, "num_tokens": 1611175.0, "completions/mean_length": 84.125, "completions/min_length": 78.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.125, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9330159425735474, "rewards/meter/std": 0.04698435589671135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9330159425735474, "rewards/total_composite/std": 0.04698435589671135, "reward": 0.9330159425735474, "reward_std": 0.04698435962200165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02119125984609127, "sampling/sampling_logp_difference/max": 1.4188241958618164, "sampling/importance_sampling_ratio/min": 0.3314521312713623, "sampling/importance_sampling_ratio/mean": 1.0041048526763916, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13521220535039902, "clip_ratio/low_mean": 0.002890269970521331, "clip_ratio/low_min": 0.002890269970521331, "clip_ratio/high_mean": 0.010773762594908476, "clip_ratio/high_max": 0.010773762594908476, "clip_ratio/region_mean": 0.013664032565429807, "reward_total_mean": 0.9330159425735474, "reward_meter_mean": 0.9330159425735474, "reward_meter_std": 0.04698435589671135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9330159425735474, "reward_total_composite_std": 0.04698435589671135, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1563.0} {"timestamp_utc": "2026-04-11T22:48:11Z", "mode": "train", "global_step": 731, "epoch": 0.02936096718480138, "loss": -0.2761, "grad_norm": 3.5135695934295654, "learning_rate": 7.787878787878789e-06, "num_tokens": 1613214.0, "completions/mean_length": 506.875, "completions/min_length": 471.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.875, "completions/mean_terminated_length": 471.0, "completions/min_terminated_length": 471.0, "completions/max_terminated_length": 471.0, "rewards/meter/mean": 0.9958064556121826, "rewards/meter/std": 0.0007912082364782691, "rewards/count_adherence/mean": 0.5277777910232544, "rewards/count_adherence/std": 0.051434461027383804, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.11126373708248138, "rewards/repeat_penalty/std": 0.07416856288909912, "rewards/total_composite/mean": 0.055620092898607254, "rewards/total_composite/std": 0.029501251876354218, "reward": 0.055620092898607254, "reward_std": 0.02950124815106392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013655466958880424, "sampling/sampling_logp_difference/max": 1.3458614349365234, "sampling/importance_sampling_ratio/min": 0.2603153586387634, "sampling/importance_sampling_ratio/mean": 0.9985577464103699, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.005113576073199511, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0005307855899445713, "clip_ratio/high_max": 0.0005307855899445713, "clip_ratio/region_mean": 0.0005307855899445713, "reward_total_mean": 0.055620092898607254, "reward_meter_mean": 0.9958064556121826, "reward_meter_std": 0.0007912082364782691, "reward_count_adherence_mean": 0.5277777910232544, "reward_count_adherence_std": 0.051434461027383804, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.11126373708248138, "reward_repeat_penalty_std": 0.07416856288909912, "reward_total_composite_mean": 0.055620092898607254, "reward_total_composite_std": 0.029501251876354218, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1564.0} {"timestamp_utc": "2026-04-11T22:48:21Z", "mode": "train", "global_step": 732, "epoch": 0.029401132666586335, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.784848484848484e-06, "num_tokens": 1614982.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9962900280952454, "rewards/meter/std": 0.001235451316460967, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0345032773911953, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.19743433594703674, "rewards/repeat_penalty/std": 0.17679214477539062, "rewards/total_composite/mean": 0.18774111568927765, "rewards/total_composite/std": 0.16288606822490692, "reward": 0.18774111568927765, "reward_std": 0.16288605332374573, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.18774111568927765, "reward_meter_mean": 0.9962900280952454, "reward_meter_std": 0.001235451316460967, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0345032773911953, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.19743433594703674, "reward_repeat_penalty_std": 0.17679214477539062, "reward_total_composite_mean": 0.18774111568927765, "reward_total_composite_std": 0.16288606822490692, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1565.0} {"timestamp_utc": "2026-04-11T22:48:26Z", "mode": "train", "global_step": 733, "epoch": 0.02944129814837129, "loss": 0.026, "grad_norm": 5.411409378051758, "learning_rate": 7.781818181818183e-06, "num_tokens": 1616806.0, "completions/mean_length": 72.0, "completions/min_length": 63.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.6948316097259521, "rewards/meter/std": 0.2794814109802246, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.5835638046264648, "rewards/total_composite/std": 0.28273841738700867, "reward": 0.5835638046264648, "reward_std": 0.28273844718933105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.055757418274879456, "sampling/sampling_logp_difference/max": 3.232696771621704, "sampling/importance_sampling_ratio/min": 0.03945096582174301, "sampling/importance_sampling_ratio/mean": 1.0083332061767578, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3269932046532631, "clip_ratio/low_mean": 0.034895967692136765, "clip_ratio/low_min": 0.034895967692136765, "clip_ratio/high_mean": 0.027756539173424244, "clip_ratio/high_max": 0.027756539173424244, "clip_ratio/region_mean": 0.06265250686556101, "reward_total_mean": 0.5835638046264648, "reward_meter_mean": 0.6948316097259521, "reward_meter_std": 0.2794814109802246, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.5835638046264648, "reward_total_composite_std": 0.28273841738700867, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1566.0} {"timestamp_utc": "2026-04-11T22:48:31Z", "mode": "train", "global_step": 734, "epoch": 0.029481463630156243, "loss": 0.0012, "grad_norm": 1.1426414251327515, "learning_rate": 7.778787878787879e-06, "num_tokens": 1619174.0, "completions/mean_length": 116.0, "completions/min_length": 116.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.0, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9807324409484863, "rewards/meter/std": 0.005540105979889631, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8406277894973755, "rewards/total_composite/std": 0.004748670384287834, "reward": 0.8406277894973755, "reward_std": 0.004748655948787928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004732904955744743, "sampling/sampling_logp_difference/max": 0.8726806640625, "sampling/importance_sampling_ratio/min": 0.4178299903869629, "sampling/importance_sampling_ratio/mean": 1.0012872219085693, "sampling/importance_sampling_ratio/max": 1.4202854633331299, "entropy": 0.025226276833564043, "clip_ratio/low_mean": 0.0021551724057644606, "clip_ratio/low_min": 0.0021551724057644606, "clip_ratio/high_mean": 0.0010775862028822303, "clip_ratio/high_max": 0.0010775862028822303, "clip_ratio/region_mean": 0.003232758608646691, "reward_total_mean": 0.8406277894973755, "reward_meter_mean": 0.9807324409484863, "reward_meter_std": 0.005540105979889631, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8406277894973755, "reward_total_composite_std": 0.004748670384287834, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1567.0} {"timestamp_utc": "2026-04-11T22:48:36Z", "mode": "train", "global_step": 735, "epoch": 0.029521629111941197, "loss": 0.0021, "grad_norm": 6.164267539978027, "learning_rate": 7.775757575757576e-06, "num_tokens": 1620938.0, "completions/mean_length": 77.5, "completions/min_length": 74.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.8432132005691528, "rewards/meter/std": 0.11800947040319443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.7734056115150452, "rewards/total_composite/std": 0.09497449547052383, "reward": 0.7734056115150452, "reward_std": 0.09497448056936264, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04740333557128906, "sampling/sampling_logp_difference/max": 1.3389692306518555, "sampling/importance_sampling_ratio/min": 0.2621157169342041, "sampling/importance_sampling_ratio/mean": 1.0057860612869263, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2497086301445961, "clip_ratio/low_mean": 0.026082158088684082, "clip_ratio/low_min": 0.026082158088684082, "clip_ratio/high_mean": 0.011470985249616206, "clip_ratio/high_max": 0.011470985249616206, "clip_ratio/region_mean": 0.03755314333830029, "reward_total_mean": 0.7734056115150452, "reward_meter_mean": 0.8432132005691528, "reward_meter_std": 0.11800947040319443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.7734056115150452, "reward_total_composite_std": 0.09497449547052383, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1568.0} {"timestamp_utc": "2026-04-11T22:48:44Z", "mode": "train", "global_step": 736, "epoch": 0.02956179459372615, "loss": 0.46, "grad_norm": 5.787602424621582, "learning_rate": 7.772727272727273e-06, "num_tokens": 1623113.0, "completions/mean_length": 115.875, "completions/min_length": 72.0, "completions/max_length": 353.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 353.0, "rewards/meter/mean": 0.7237052917480469, "rewards/meter/std": 0.3117098808288574, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6369646191596985, "rewards/total_composite/std": 0.40405309200286865, "reward": 0.6369646191596985, "reward_std": 0.40405309200286865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08920754492282867, "sampling/sampling_logp_difference/max": 1.3201618194580078, "sampling/importance_sampling_ratio/min": 0.2670920789241791, "sampling/importance_sampling_ratio/mean": 1.0180284976959229, "sampling/importance_sampling_ratio/max": 1.964148998260498, "entropy": 1.2863622568547726, "clip_ratio/low_mean": 0.008429729146882892, "clip_ratio/low_min": 0.008429729146882892, "clip_ratio/high_mean": 0.01677176496013999, "clip_ratio/high_max": 0.01677176496013999, "clip_ratio/region_mean": 0.02520149410702288, "reward_total_mean": 0.6369646191596985, "reward_meter_mean": 0.7237052917480469, "reward_meter_std": 0.3117098808288574, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6369646191596985, "reward_total_composite_std": 0.40405309200286865, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1569.0} {"timestamp_utc": "2026-04-11T22:48:54Z", "mode": "train", "global_step": 737, "epoch": 0.029601960075511104, "loss": -0.2132, "grad_norm": 1.1047923564910889, "learning_rate": 7.76969696969697e-06, "num_tokens": 1625144.0, "completions/mean_length": 294.875, "completions/min_length": 147.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 164.60000610351562, "completions/min_terminated_length": 147.0, "completions/max_terminated_length": 191.0, "rewards/meter/mean": 0.3803099989891052, "rewards/meter/std": 0.2973701059818268, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.22160132229328156, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/repeat_penalty/mean": 0.797619104385376, "rewards/repeat_penalty/std": 0.2185886949300766, "rewards/total_composite/mean": 0.20652014017105103, "rewards/total_composite/std": 0.19584397971630096, "reward": 0.20652014017105103, "reward_std": 0.19584397971630096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030263110995292664, "sampling/sampling_logp_difference/max": 1.4690535068511963, "sampling/importance_sampling_ratio/min": 0.2301432192325592, "sampling/importance_sampling_ratio/mean": 1.0015567541122437, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12422919739037752, "clip_ratio/low_mean": 0.0008333333535119891, "clip_ratio/low_min": 0.0008333333535119891, "clip_ratio/high_mean": 0.015350477071478963, "clip_ratio/high_max": 0.015350477071478963, "clip_ratio/region_mean": 0.016183810424990952, "reward_total_mean": 0.20652014017105103, "reward_meter_mean": 0.3803099989891052, "reward_meter_std": 0.2973701059818268, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.22160132229328156, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_repeat_penalty_mean": 0.797619104385376, "reward_repeat_penalty_std": 0.2185886949300766, "reward_total_composite_mean": 0.20652014017105103, "reward_total_composite_std": 0.19584397971630096, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1570.0} {"timestamp_utc": "2026-04-11T22:49:04Z", "mode": "train", "global_step": 738, "epoch": 0.02964212555729606, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.766666666666666e-06, "num_tokens": 1626768.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9730854630470276, "rewards/meter/std": 0.03588658943772316, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.050507619976997375, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.2781907618045807, "rewards/repeat_penalty/std": 0.2387264519929886, "rewards/total_composite/mean": 0.2381971776485443, "rewards/total_composite/std": 0.20977550745010376, "reward": 0.2381971776485443, "reward_std": 0.20977550745010376, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.2381971776485443, "reward_meter_mean": 0.9730854630470276, "reward_meter_std": 0.03588658943772316, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.050507619976997375, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.2781907618045807, "reward_repeat_penalty_std": 0.2387264519929886, "reward_total_composite_mean": 0.2381971776485443, "reward_total_composite_std": 0.20977550745010376, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1571.0} {"timestamp_utc": "2026-04-11T22:49:13Z", "mode": "train", "global_step": 739, "epoch": 0.029682291039081012, "loss": 0.0363, "grad_norm": 1.4499318599700928, "learning_rate": 7.763636363636364e-06, "num_tokens": 1631595.0, "completions/mean_length": 423.375, "completions/min_length": 398.0, "completions/max_length": 483.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 423.375, "completions/min_terminated_length": 398.0, "completions/max_terminated_length": 483.0, "rewards/meter/mean": 0.7534134387969971, "rewards/meter/std": 0.2928631901741028, "rewards/count_adherence/mean": 0.3888888955116272, "rewards/count_adherence/std": 0.059391383081674576, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5623973608016968, "rewards/repeat_penalty/std": 0.18699996173381805, "rewards/total_composite/mean": 0.15555939078330994, "rewards/total_composite/std": 0.07782954722642899, "reward": 0.15555939078330994, "reward_std": 0.07782954722642899, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013467294164001942, "sampling/sampling_logp_difference/max": 3.2660930156707764, "sampling/importance_sampling_ratio/min": 0.0381552055478096, "sampling/importance_sampling_ratio/mean": 0.9997313022613525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0637149391695857, "clip_ratio/low_mean": 0.005539899080758914, "clip_ratio/low_min": 0.005539899080758914, "clip_ratio/high_mean": 0.004910050658509135, "clip_ratio/high_max": 0.004910050658509135, "clip_ratio/region_mean": 0.01044994973926805, "reward_total_mean": 0.15555939078330994, "reward_meter_mean": 0.7534134387969971, "reward_meter_std": 0.2928631901741028, "reward_count_adherence_mean": 0.3888888955116272, "reward_count_adherence_std": 0.059391383081674576, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5623973608016968, "reward_repeat_penalty_std": 0.18699996173381805, "reward_total_composite_mean": 0.15555939078330994, "reward_total_composite_std": 0.07782954722642899, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1572.0} {"timestamp_utc": "2026-04-11T22:49:19Z", "mode": "train", "global_step": 740, "epoch": 0.029722456520865966, "loss": -0.0085, "grad_norm": 3.1338510513305664, "learning_rate": 7.76060606060606e-06, "num_tokens": 1634096.0, "completions/mean_length": 133.625, "completions/min_length": 126.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.625, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.11201652884483337, "rewards/meter/std": 0.19264455139636993, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6964285969734192, "rewards/repeat_penalty/std": 0.11921756714582443, "rewards/total_composite/mean": 0.0800618976354599, "rewards/total_composite/std": 0.13757191598415375, "reward": 0.0800618976354599, "reward_std": 0.13757191598415375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02300534024834633, "sampling/sampling_logp_difference/max": 1.0942306518554688, "sampling/importance_sampling_ratio/min": 0.33479708433151245, "sampling/importance_sampling_ratio/mean": 1.005319595336914, "sampling/importance_sampling_ratio/max": 1.8425383567810059, "entropy": 0.15879025869071484, "clip_ratio/low_mean": 0.008456611772999167, "clip_ratio/low_min": 0.008456611772999167, "clip_ratio/high_mean": 0.009530608775094151, "clip_ratio/high_max": 0.009530608775094151, "clip_ratio/region_mean": 0.01798722054809332, "reward_total_mean": 0.0800618976354599, "reward_meter_mean": 0.11201652884483337, "reward_meter_std": 0.19264455139636993, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6964285969734192, "reward_repeat_penalty_std": 0.11921756714582443, "reward_total_composite_mean": 0.0800618976354599, "reward_total_composite_std": 0.13757191598415375, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1573.0} {"timestamp_utc": "2026-04-11T22:49:24Z", "mode": "train", "global_step": 741, "epoch": 0.02976262200265092, "loss": -0.0061, "grad_norm": 1.2720859050750732, "learning_rate": 7.757575757575758e-06, "num_tokens": 1636098.0, "completions/mean_length": 79.25, "completions/min_length": 78.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.25, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9925205707550049, "rewards/meter/std": 0.0022588411811739206, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.794016420841217, "rewards/total_composite/std": 0.001807067426852882, "reward": 0.794016420841217, "reward_std": 0.0018070697551593184, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011411315761506557, "sampling/sampling_logp_difference/max": 0.8109602928161621, "sampling/importance_sampling_ratio/min": 0.4444310963153839, "sampling/importance_sampling_ratio/mean": 1.001147985458374, "sampling/importance_sampling_ratio/max": 1.5393544435501099, "entropy": 0.07466331776231527, "clip_ratio/low_mean": 0.006329114083200693, "clip_ratio/low_min": 0.006329114083200693, "clip_ratio/high_mean": 0.007794186705723405, "clip_ratio/high_max": 0.007794186705723405, "clip_ratio/region_mean": 0.014123300788924098, "reward_total_mean": 0.794016420841217, "reward_meter_mean": 0.9925205707550049, "reward_meter_std": 0.0022588411811739206, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.794016420841217, "reward_total_composite_std": 0.001807067426852882, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1574.0} {"timestamp_utc": "2026-04-11T22:49:29Z", "mode": "train", "global_step": 742, "epoch": 0.029802787484435874, "loss": 0.012, "grad_norm": 7.301284313201904, "learning_rate": 7.754545454545455e-06, "num_tokens": 1638424.0, "completions/mean_length": 120.75, "completions/min_length": 114.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.75, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.7745586633682251, "rewards/meter/std": 0.25043216347694397, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6964285969734192, "rewards/repeat_penalty/std": 0.17806050181388855, "rewards/total_composite/mean": 0.5217785835266113, "rewards/total_composite/std": 0.1874629557132721, "reward": 0.5217785835266113, "reward_std": 0.1874629259109497, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04599983990192413, "sampling/sampling_logp_difference/max": 4.404999732971191, "sampling/importance_sampling_ratio/min": 0.012216109782457352, "sampling/importance_sampling_ratio/mean": 0.9962561726570129, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12627906422130764, "clip_ratio/low_mean": 0.021951976465061307, "clip_ratio/low_min": 0.021951976465061307, "clip_ratio/high_mean": 0.018717249389737844, "clip_ratio/high_max": 0.018717249389737844, "clip_ratio/region_mean": 0.04066922585479915, "reward_total_mean": 0.5217785835266113, "reward_meter_mean": 0.7745586633682251, "reward_meter_std": 0.25043216347694397, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6964285969734192, "reward_repeat_penalty_std": 0.17806050181388855, "reward_total_composite_mean": 0.5217785835266113, "reward_total_composite_std": 0.1874629557132721, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1575.0} {"timestamp_utc": "2026-04-11T22:49:39Z", "mode": "train", "global_step": 743, "epoch": 0.02984295296622083, "loss": 0.1769, "grad_norm": 1.0547409057617188, "learning_rate": 7.751515151515153e-06, "num_tokens": 1642208.0, "completions/mean_length": 510.0, "completions/min_length": 507.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 508.0, "completions/min_terminated_length": 507.0, "completions/max_terminated_length": 510.0, "rewards/meter/mean": 0.45572322607040405, "rewards/meter/std": 0.1519794911146164, "rewards/count_adherence/mean": 0.6153846383094788, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5480158925056458, "rewards/repeat_penalty/std": 0.010451768524944782, "rewards/total_composite/mean": 0.13946115970611572, "rewards/total_composite/std": 0.07740650326013565, "reward": 0.13946115970611572, "reward_std": 0.07740650326013565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00907099712640047, "sampling/sampling_logp_difference/max": 1.135289192199707, "sampling/importance_sampling_ratio/min": 0.32132917642593384, "sampling/importance_sampling_ratio/mean": 1.0025880336761475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03308352828025818, "clip_ratio/low_mean": 0.0017248676158487797, "clip_ratio/low_min": 0.0017248676158487797, "clip_ratio/high_mean": 0.0012283907853998244, "clip_ratio/high_max": 0.0012283907853998244, "clip_ratio/region_mean": 0.002953258401248604, "reward_total_mean": 0.13946115970611572, "reward_meter_mean": 0.45572322607040405, "reward_meter_std": 0.1519794911146164, "reward_count_adherence_mean": 0.6153846383094788, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5480158925056458, "reward_repeat_penalty_std": 0.010451768524944782, "reward_total_composite_mean": 0.13946115970611572, "reward_total_composite_std": 0.07740650326013565, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1576.0} {"timestamp_utc": "2026-04-11T22:49:44Z", "mode": "train", "global_step": 744, "epoch": 0.029883118448005785, "loss": 0.0283, "grad_norm": 6.8046875, "learning_rate": 7.74848484848485e-06, "num_tokens": 1643996.0, "completions/mean_length": 64.5, "completions/min_length": 61.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.5396110415458679, "rewards/meter/std": 0.20649512112140656, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5396110415458679, "rewards/total_composite/std": 0.20649512112140656, "reward": 0.5396110415458679, "reward_std": 0.20649512112140656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06342907249927521, "sampling/sampling_logp_difference/max": 1.2414631843566895, "sampling/importance_sampling_ratio/min": 0.28896111249923706, "sampling/importance_sampling_ratio/mean": 1.0097684860229492, "sampling/importance_sampling_ratio/max": 1.632293462753296, "entropy": 0.6070369817316532, "clip_ratio/low_mean": 0.02111415727995336, "clip_ratio/low_min": 0.02111415727995336, "clip_ratio/high_mean": 0.027501578675583005, "clip_ratio/high_max": 0.027501578675583005, "clip_ratio/region_mean": 0.048615735955536366, "reward_total_mean": 0.5396110415458679, "reward_meter_mean": 0.5396110415458679, "reward_meter_std": 0.20649512112140656, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5396110415458679, "reward_total_composite_std": 0.20649512112140656, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1577.0} {"timestamp_utc": "2026-04-11T22:49:49Z", "mode": "train", "global_step": 745, "epoch": 0.02992328392979074, "loss": 0.0099, "grad_norm": 9.705404281616211, "learning_rate": 7.745454545454545e-06, "num_tokens": 1645756.0, "completions/mean_length": 63.0, "completions/min_length": 62.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9103853106498718, "rewards/meter/std": 0.22165298461914062, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9103853106498718, "rewards/total_composite/std": 0.22165298461914062, "reward": 0.9103853106498718, "reward_std": 0.22165298461914062, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07189808040857315, "sampling/sampling_logp_difference/max": 2.013796806335449, "sampling/importance_sampling_ratio/min": 0.13348090648651123, "sampling/importance_sampling_ratio/mean": 1.0033738613128662, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25149067025631666, "clip_ratio/low_mean": 0.008064515888690948, "clip_ratio/low_min": 0.008064515888690948, "clip_ratio/high_mean": 0.043645198456943035, "clip_ratio/high_max": 0.043645198456943035, "clip_ratio/region_mean": 0.051709714345633984, "reward_total_mean": 0.9103853106498718, "reward_meter_mean": 0.9103853106498718, "reward_meter_std": 0.22165298461914062, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9103853106498718, "reward_total_composite_std": 0.22165298461914062, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1578.0} {"timestamp_utc": "2026-04-11T22:49:56Z", "mode": "train", "global_step": 746, "epoch": 0.029963449411575693, "loss": 0.0227, "grad_norm": 2.2934165000915527, "learning_rate": 7.742424242424244e-06, "num_tokens": 1649182.0, "completions/mean_length": 238.25, "completions/min_length": 205.0, "completions/max_length": 260.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 238.25, "completions/min_terminated_length": 205.0, "completions/max_terminated_length": 260.0, "rewards/meter/mean": 0.8676279783248901, "rewards/meter/std": 0.3215867578983307, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.36800700426101685, "rewards/repeat_penalty/std": 0.24577626585960388, "rewards/total_composite/mean": 0.2228499948978424, "rewards/total_composite/std": 0.18318407237529755, "reward": 0.2228499948978424, "reward_std": 0.18318407237529755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018860070034861565, "sampling/sampling_logp_difference/max": 4.0910139083862305, "sampling/importance_sampling_ratio/min": 0.016722269356250763, "sampling/importance_sampling_ratio/mean": 1.0014268159866333, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06980151077732444, "clip_ratio/low_mean": 0.003769519622437656, "clip_ratio/low_min": 0.003769519622437656, "clip_ratio/high_mean": 0.006430621142499149, "clip_ratio/high_max": 0.006430621142499149, "clip_ratio/region_mean": 0.010200140764936805, "reward_total_mean": 0.2228499948978424, "reward_meter_mean": 0.8676279783248901, "reward_meter_std": 0.3215867578983307, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.36800700426101685, "reward_repeat_penalty_std": 0.24577626585960388, "reward_total_composite_mean": 0.2228499948978424, "reward_total_composite_std": 0.18318407237529755, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1579.0} {"timestamp_utc": "2026-04-11T22:50:00Z", "mode": "train", "global_step": 747, "epoch": 0.030003614893360647, "loss": -0.0252, "grad_norm": 7.876811981201172, "learning_rate": 7.73939393939394e-06, "num_tokens": 1650683.0, "completions/mean_length": 40.625, "completions/min_length": 38.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9663740396499634, "rewards/meter/std": 0.011263493448495865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9663740396499634, "rewards/total_composite/std": 0.011263493448495865, "reward": 0.9663740396499634, "reward_std": 0.011263499036431313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03788968175649643, "sampling/sampling_logp_difference/max": 0.8854336738586426, "sampling/importance_sampling_ratio/min": 0.41253525018692017, "sampling/importance_sampling_ratio/mean": 1.0029720067977905, "sampling/importance_sampling_ratio/max": 1.344473123550415, "entropy": 0.28702736645936966, "clip_ratio/low_mean": 0.012351190904155374, "clip_ratio/low_min": 0.012351190904155374, "clip_ratio/high_mean": 0.018227352295070887, "clip_ratio/high_max": 0.018227352295070887, "clip_ratio/region_mean": 0.03057854319922626, "reward_total_mean": 0.9663740396499634, "reward_meter_mean": 0.9663740396499634, "reward_meter_std": 0.011263493448495865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9663740396499634, "reward_total_composite_std": 0.011263493448495865, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1580.0} {"timestamp_utc": "2026-04-11T22:50:05Z", "mode": "train", "global_step": 748, "epoch": 0.0300437803751456, "loss": 0.031, "grad_norm": 4.674757957458496, "learning_rate": 7.736363636363637e-06, "num_tokens": 1652564.0, "completions/mean_length": 82.125, "completions/min_length": 74.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.125, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.8491703271865845, "rewards/meter/std": 0.15907908976078033, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8491703271865845, "rewards/total_composite/std": 0.15907908976078033, "reward": 0.8491703271865845, "reward_std": 0.15907907485961914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0295439250767231, "sampling/sampling_logp_difference/max": 1.300436019897461, "sampling/importance_sampling_ratio/min": 0.2724129855632782, "sampling/importance_sampling_ratio/mean": 1.0036686658859253, "sampling/importance_sampling_ratio/max": 1.843922734260559, "entropy": 0.22462552785873413, "clip_ratio/low_mean": 0.002961171790957451, "clip_ratio/low_min": 0.002961171790957451, "clip_ratio/high_mean": 0.021474606008268893, "clip_ratio/high_max": 0.021474606008268893, "clip_ratio/region_mean": 0.024435777799226344, "reward_total_mean": 0.8491703271865845, "reward_meter_mean": 0.8491703271865845, "reward_meter_std": 0.15907908976078033, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8491703271865845, "reward_total_composite_std": 0.15907908976078033, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1581.0} {"timestamp_utc": "2026-04-11T22:50:10Z", "mode": "train", "global_step": 749, "epoch": 0.030083945856930555, "loss": 0.0372, "grad_norm": 6.799373626708984, "learning_rate": 7.733333333333334e-06, "num_tokens": 1654625.0, "completions/mean_length": 93.625, "completions/min_length": 87.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.625, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9564855694770813, "rewards/meter/std": 0.10235674679279327, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8075714707374573, "rewards/total_composite/std": 0.0810769721865654, "reward": 0.8075714707374573, "reward_std": 0.0810769572854042, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03427453711628914, "sampling/sampling_logp_difference/max": 2.4858551025390625, "sampling/importance_sampling_ratio/min": 0.08325432986021042, "sampling/importance_sampling_ratio/mean": 0.9965777397155762, "sampling/importance_sampling_ratio/max": 1.903843641281128, "entropy": 0.12411046819761395, "clip_ratio/low_mean": 0.026110241888090968, "clip_ratio/low_min": 0.026110241888090968, "clip_ratio/high_mean": 0.008620689623057842, "clip_ratio/high_max": 0.008620689623057842, "clip_ratio/region_mean": 0.03473093151114881, "reward_total_mean": 0.8075714707374573, "reward_meter_mean": 0.9564855694770813, "reward_meter_std": 0.10235674679279327, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.8075714707374573, "reward_total_composite_std": 0.0810769721865654, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1582.0} {"timestamp_utc": "2026-04-11T22:50:15Z", "mode": "train", "global_step": 750, "epoch": 0.03012411133871551, "loss": 0.031, "grad_norm": 6.688387870788574, "learning_rate": 7.730303030303032e-06, "num_tokens": 1656513.0, "completions/mean_length": 67.0, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9406866431236267, "rewards/meter/std": 0.1574798822402954, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.8990964889526367, "rewards/total_composite/std": 0.18213681876659393, "reward": 0.8990964889526367, "reward_std": 0.18213684856891632, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035663776099681854, "sampling/sampling_logp_difference/max": 0.9646925926208496, "sampling/importance_sampling_ratio/min": 0.42169255018234253, "sampling/importance_sampling_ratio/mean": 1.0121128559112549, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21013184823095798, "clip_ratio/low_mean": 0.00892857147846371, "clip_ratio/low_min": 0.00892857147846371, "clip_ratio/high_mean": 0.022817256744019687, "clip_ratio/high_max": 0.022817256744019687, "clip_ratio/region_mean": 0.031745828222483397, "reward_total_mean": 0.8990964889526367, "reward_meter_mean": 0.9406866431236267, "reward_meter_std": 0.1574798822402954, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.8990964889526367, "reward_total_composite_std": 0.18213681876659393, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1583.0} {"timestamp_utc": "2026-04-11T22:51:48Z", "mode": "eval", "global_step": 750, "epoch": 0.03012411133871551, "eval_loss": NaN, "eval_runtime": 93.1633, "eval_samples_per_second": 1.116, "eval_steps_per_second": 0.14, "eval_num_tokens": 1656513.0, "eval_completions/mean_length": 284.52884615384613, "eval_completions/min_length": 61.46153846153846, "eval_completions/max_length": 496.7692307692308, "eval_completions/clipped_ratio": 0.23076923076923078, "eval_completions/mean_terminated_length": 215.78288092980017, "eval_completions/min_terminated_length": 61.46153846153846, "eval_completions/max_terminated_length": 393.46153846153845, "eval_rewards/meter/mean": 0.6660121427132533, "eval_rewards/meter/std": 0.37975076070198643, "eval_rewards/count_adherence/mean": 0.7424245018225449, "eval_rewards/count_adherence/std": 0.2582179422561939, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.08158924258672275, "eval_rewards/repeat_penalty/mean": 0.638540084545429, "eval_rewards/repeat_penalty/std": 0.270676647241299, "eval_rewards/total_composite/mean": 0.33899185634576356, "eval_rewards/total_composite/std": 0.3308297275350644, "eval_reward": 0.33899185634576356, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.012646582407447008, "eval_sampling/sampling_logp_difference/max": 0.8113565261547382, "eval_sampling/importance_sampling_ratio/min": 0.4615517258644104, "eval_sampling/importance_sampling_ratio/mean": 1.0036690326837392, "eval_sampling/importance_sampling_ratio/max": 1.3942875678722675, "eval_entropy": 0.17351046003974402, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.33899185634576356, "eval_reward_meter_mean": 0.6660121427132533, "eval_reward_meter_std": 0.37975076070198643, "eval_reward_count_adherence_mean": 0.7424245018225449, "eval_reward_count_adherence_std": 0.2582179422561939, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.08158924258672275, "eval_reward_repeat_penalty_mean": 0.638540084545429, "eval_reward_repeat_penalty_std": 0.270676647241299, "eval_reward_total_composite_mean": 0.33899185634576356, "eval_reward_total_composite_std": 0.3308297275350644, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1583.0} {"timestamp_utc": "2026-04-11T22:51:55Z", "mode": "train", "global_step": 751, "epoch": 0.030164276820500463, "loss": 0.0024, "grad_norm": 5.31432580947876, "learning_rate": 7.727272727272727e-06, "num_tokens": 1658395.0, "completions/mean_length": 59.25, "completions/min_length": 59.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9893741607666016, "rewards/meter/std": 0.007569636683911085, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9893741607666016, "rewards/total_composite/std": 0.007569636683911085, "reward": 0.9893741607666016, "reward_std": 0.007569642271846533, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013296845369040966, "sampling/sampling_logp_difference/max": 0.9290802478790283, "sampling/importance_sampling_ratio/min": 0.3949167728424072, "sampling/importance_sampling_ratio/mean": 1.0002750158309937, "sampling/importance_sampling_ratio/max": 1.868045449256897, "entropy": 0.052004152443259954, "clip_ratio/low_mean": 0.006320621585473418, "clip_ratio/low_min": 0.006320621585473418, "clip_ratio/high_mean": 0.008403955027461052, "clip_ratio/high_max": 0.008403955027461052, "clip_ratio/region_mean": 0.01472457661293447, "reward_total_mean": 0.9893741607666016, "reward_meter_mean": 0.9893741607666016, "reward_meter_std": 0.007569636683911085, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9893741607666016, "reward_total_composite_std": 0.007569636683911085, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1584.0} {"timestamp_utc": "2026-04-11T22:52:00Z", "mode": "train", "global_step": 752, "epoch": 0.030204442302285417, "loss": 0.0051, "grad_norm": 4.07772970199585, "learning_rate": 7.724242424242424e-06, "num_tokens": 1660117.0, "completions/mean_length": 59.25, "completions/min_length": 59.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9949396848678589, "rewards/meter/std": 0.00028474157443270087, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9534660577774048, "rewards/total_composite/std": 0.11713220924139023, "reward": 0.9534660577774048, "reward_std": 0.11713218688964844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009837880730628967, "sampling/sampling_logp_difference/max": 1.2378616333007812, "sampling/importance_sampling_ratio/min": 0.2900037169456482, "sampling/importance_sampling_ratio/mean": 1.0019704103469849, "sampling/importance_sampling_ratio/max": 1.3358639478683472, "entropy": 0.03427604655735195, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/high_mean": 0.006320621585473418, "clip_ratio/high_max": 0.006320621585473418, "clip_ratio/region_mean": 0.008403955027461052, "reward_total_mean": 0.9534660577774048, "reward_meter_mean": 0.9949396848678589, "reward_meter_std": 0.00028474157443270087, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9534660577774048, "reward_total_composite_std": 0.11713220924139023, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1585.0} {"timestamp_utc": "2026-04-11T22:52:04Z", "mode": "train", "global_step": 753, "epoch": 0.03024460778407037, "loss": -0.0196, "grad_norm": 14.107855796813965, "learning_rate": 7.721212121212122e-06, "num_tokens": 1661428.0, "completions/mean_length": 35.875, "completions/min_length": 34.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9349486827850342, "rewards/meter/std": 0.1060987114906311, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9349486827850342, "rewards/total_composite/std": 0.1060987114906311, "reward": 0.9349486827850342, "reward_std": 0.1060987114906311, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07551710307598114, "sampling/sampling_logp_difference/max": 1.5768651962280273, "sampling/importance_sampling_ratio/min": 0.20662181079387665, "sampling/importance_sampling_ratio/mean": 1.0009469985961914, "sampling/importance_sampling_ratio/max": 1.6115726232528687, "entropy": 0.6193088293075562, "clip_ratio/low_mean": 0.017439668532460928, "clip_ratio/low_min": 0.017439668532460928, "clip_ratio/high_mean": 0.04496192745864391, "clip_ratio/high_max": 0.04496192745864391, "clip_ratio/region_mean": 0.06240159599110484, "reward_total_mean": 0.9349486827850342, "reward_meter_mean": 0.9349486827850342, "reward_meter_std": 0.1060987114906311, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9349486827850342, "reward_total_composite_std": 0.1060987114906311, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1586.0} {"timestamp_utc": "2026-04-11T22:52:09Z", "mode": "train", "global_step": 754, "epoch": 0.030284773265855325, "loss": 0.0091, "grad_norm": 2.418649673461914, "learning_rate": 7.718181818181819e-06, "num_tokens": 1663080.0, "completions/mean_length": 41.5, "completions/min_length": 41.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9943797588348389, "rewards/meter/std": 0.0011224248446524143, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943797588348389, "rewards/total_composite/std": 0.0011224248446524143, "reward": 0.9943797588348389, "reward_std": 0.0011224271729588509, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030370449647307396, "sampling/sampling_logp_difference/max": 1.4918346405029297, "sampling/importance_sampling_ratio/min": 0.2249595671892166, "sampling/importance_sampling_ratio/mean": 0.9925946593284607, "sampling/importance_sampling_ratio/max": 1.2741270065307617, "entropy": 0.11915541160851717, "clip_ratio/low_mean": 0.008928571594879031, "clip_ratio/low_min": 0.008928571594879031, "clip_ratio/high_mean": 0.018074912950396538, "clip_ratio/high_max": 0.018074912950396538, "clip_ratio/region_mean": 0.02700348454527557, "reward_total_mean": 0.9943797588348389, "reward_meter_mean": 0.9943797588348389, "reward_meter_std": 0.0011224248446524143, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943797588348389, "reward_total_composite_std": 0.0011224248446524143, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1587.0} {"timestamp_utc": "2026-04-11T22:52:13Z", "mode": "train", "global_step": 755, "epoch": 0.03032493874764028, "loss": -0.0018, "grad_norm": 5.144253730773926, "learning_rate": 7.715151515151516e-06, "num_tokens": 1664616.0, "completions/mean_length": 41.0, "completions/min_length": 41.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9944278001785278, "rewards/meter/std": 0.0012945194030180573, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9944278001785278, "rewards/total_composite/std": 0.0012945194030180573, "reward": 0.9944278001785278, "reward_std": 0.0012945224298164248, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014835118316113949, "sampling/sampling_logp_difference/max": 0.617079496383667, "sampling/importance_sampling_ratio/min": 0.5395178198814392, "sampling/importance_sampling_ratio/mean": 1.0021907091140747, "sampling/importance_sampling_ratio/max": 1.2102103233337402, "entropy": 0.10165042616426945, "clip_ratio/low_mean": 0.0030487803742289543, "clip_ratio/low_min": 0.0030487803742289543, "clip_ratio/high_mean": 0.012195121496915817, "clip_ratio/high_max": 0.012195121496915817, "clip_ratio/region_mean": 0.015243901871144772, "reward_total_mean": 0.9944278001785278, "reward_meter_mean": 0.9944278001785278, "reward_meter_std": 0.0012945194030180573, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9944278001785278, "reward_total_composite_std": 0.0012945194030180573, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1588.0} {"timestamp_utc": "2026-04-11T22:52:19Z", "mode": "train", "global_step": 756, "epoch": 0.030365104229425233, "loss": -0.0099, "grad_norm": 2.4242031574249268, "learning_rate": 7.712121212121213e-06, "num_tokens": 1666878.0, "completions/mean_length": 113.75, "completions/min_length": 111.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.75, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.9662827253341675, "rewards/meter/std": 0.009305375628173351, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6500000357627869, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.6285786032676697, "rewards/total_composite/std": 0.0941418930888176, "reward": 0.6285786032676697, "reward_std": 0.09414192289113998, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011911381967365742, "sampling/sampling_logp_difference/max": 1.6042141914367676, "sampling/importance_sampling_ratio/min": 0.20104746520519257, "sampling/importance_sampling_ratio/mean": 1.0027672052383423, "sampling/importance_sampling_ratio/max": 1.911029577255249, "entropy": 0.0643842932768166, "clip_ratio/low_mean": 0.005512091098353267, "clip_ratio/low_min": 0.005512091098353267, "clip_ratio/high_mean": 0.005401917500421405, "clip_ratio/high_max": 0.005401917500421405, "clip_ratio/region_mean": 0.010914008598774672, "reward_total_mean": 0.6285786032676697, "reward_meter_mean": 0.9662827253341675, "reward_meter_std": 0.009305375628173351, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6500000357627869, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.6285786032676697, "reward_total_composite_std": 0.0941418930888176, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1589.0} {"timestamp_utc": "2026-04-11T22:52:26Z", "mode": "train", "global_step": 757, "epoch": 0.030405269711210187, "loss": 0.0259, "grad_norm": 1.2799259424209595, "learning_rate": 7.709090909090909e-06, "num_tokens": 1670368.0, "completions/mean_length": 243.25, "completions/min_length": 217.0, "completions/max_length": 267.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 243.25, "completions/min_terminated_length": 217.0, "completions/max_terminated_length": 267.0, "rewards/meter/mean": 0.9969884157180786, "rewards/meter/std": 0.000910833477973938, "rewards/count_adherence/mean": 0.6500000357627869, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4187062978744507, "rewards/repeat_penalty/std": 0.23771634697914124, "rewards/total_composite/mean": 0.2822417914867401, "rewards/total_composite/std": 0.18207497894763947, "reward": 0.2822417914867401, "reward_std": 0.18207496404647827, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00974962953478098, "sampling/sampling_logp_difference/max": 1.2791423797607422, "sampling/importance_sampling_ratio/min": 0.27827587723731995, "sampling/importance_sampling_ratio/mean": 1.0003676414489746, "sampling/importance_sampling_ratio/max": 1.5265671014785767, "entropy": 0.06725997012108564, "clip_ratio/low_mean": 0.0010245901066809893, "clip_ratio/low_min": 0.0010245901066809893, "clip_ratio/high_mean": 0.007282168400706723, "clip_ratio/high_max": 0.007282168400706723, "clip_ratio/region_mean": 0.008306758507387713, "reward_total_mean": 0.2822417914867401, "reward_meter_mean": 0.9969884157180786, "reward_meter_std": 0.000910833477973938, "reward_count_adherence_mean": 0.6500000357627869, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4187062978744507, "reward_repeat_penalty_std": 0.23771634697914124, "reward_total_composite_mean": 0.2822417914867401, "reward_total_composite_std": 0.18207497894763947, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1590.0} {"timestamp_utc": "2026-04-11T22:52:37Z", "mode": "train", "global_step": 758, "epoch": 0.03044543519299514, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.706060606060606e-06, "num_tokens": 1672112.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.8747541308403015, "rewards/meter/std": 0.1777074635028839, "rewards/count_adherence/mean": 0.15625, "rewards/count_adherence/std": 0.11080066114664078, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.42480412125587463, "rewards/repeat_penalty/std": 0.2929826080799103, "rewards/total_composite/mean": 0.059182293713092804, "rewards/total_composite/std": 0.06593845039606094, "reward": 0.059182293713092804, "reward_std": 0.06593845039606094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.059182293713092804, "reward_meter_mean": 0.8747541308403015, "reward_meter_std": 0.1777074635028839, "reward_count_adherence_mean": 0.15625, "reward_count_adherence_std": 0.11080066114664078, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.42480412125587463, "reward_repeat_penalty_std": 0.2929826080799103, "reward_total_composite_mean": 0.059182293713092804, "reward_total_composite_std": 0.06593845039606094, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1591.0} {"timestamp_utc": "2026-04-11T22:52:42Z", "mode": "train", "global_step": 759, "epoch": 0.030485600674780094, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.703030303030304e-06, "num_tokens": 1674144.0, "completions/mean_length": 87.0, "completions/min_length": 87.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.0, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9949710369110107, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6000000238418579, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5969825983047485, "rewards/total_composite/std": 0.0, "reward": 0.5969825983047485, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.001654994674026966, "sampling/sampling_logp_difference/max": 0.1641908884048462, "sampling/importance_sampling_ratio/min": 0.848580002784729, "sampling/importance_sampling_ratio/mean": 1.0006695985794067, "sampling/importance_sampling_ratio/max": 1.0613410472869873, "entropy": 0.014668526826426387, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.5969825983047485, "reward_meter_mean": 0.9949710369110107, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6000000238418579, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5969825983047485, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1592.0} {"timestamp_utc": "2026-04-11T22:52:46Z", "mode": "train", "global_step": 760, "epoch": 0.03052576615656505, "loss": -0.0037, "grad_norm": 4.493424415588379, "learning_rate": 7.7e-06, "num_tokens": 1675945.0, "completions/mean_length": 54.125, "completions/min_length": 54.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9517734050750732, "rewards/meter/std": 0.0011382178636267781, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.6740554571151733, "rewards/total_composite/std": 0.11107677966356277, "reward": 0.6740554571151733, "reward_std": 0.11107677221298218, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005725136026740074, "sampling/sampling_logp_difference/max": 0.4777810573577881, "sampling/importance_sampling_ratio/min": 0.7561957836151123, "sampling/importance_sampling_ratio/mean": 1.0033941268920898, "sampling/importance_sampling_ratio/max": 1.612492322921753, "entropy": 0.042751661501824856, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004629629664123058, "reward_total_mean": 0.6740554571151733, "reward_meter_mean": 0.9517734050750732, "reward_meter_std": 0.0011382178636267781, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_total_composite_mean": 0.6740554571151733, "reward_total_composite_std": 0.11107677966356277, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1593.0} {"timestamp_utc": "2026-04-11T22:52:51Z", "mode": "train", "global_step": 761, "epoch": 0.030565931638350002, "loss": -0.0023, "grad_norm": 0.8903390765190125, "learning_rate": 7.696969696969696e-06, "num_tokens": 1677747.0, "completions/mean_length": 58.25, "completions/min_length": 58.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9949068427085876, "rewards/meter/std": 0.0007968654972501099, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.7047345042228699, "rewards/total_composite/std": 0.1173337996006012, "reward": 0.7047345042228699, "reward_std": 0.1173337996006012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011640225537121296, "sampling/sampling_logp_difference/max": 1.0540056228637695, "sampling/importance_sampling_ratio/min": 0.3485388159751892, "sampling/importance_sampling_ratio/mean": 1.0014004707336426, "sampling/importance_sampling_ratio/max": 1.6007080078125, "entropy": 0.05586709058843553, "clip_ratio/low_mean": 0.006428988883271813, "clip_ratio/low_min": 0.006428988883271813, "clip_ratio/high_mean": 0.0021186440717428923, "clip_ratio/high_max": 0.0021186440717428923, "clip_ratio/region_mean": 0.008547632955014706, "reward_total_mean": 0.7047345042228699, "reward_meter_mean": 0.9949068427085876, "reward_meter_std": 0.0007968654972501099, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_total_composite_mean": 0.7047345042228699, "reward_total_composite_std": 0.1173337996006012, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1594.0} {"timestamp_utc": "2026-04-11T22:52:56Z", "mode": "train", "global_step": 762, "epoch": 0.030606097120134956, "loss": 0.024, "grad_norm": 5.6463236808776855, "learning_rate": 7.693939393939395e-06, "num_tokens": 1679613.0, "completions/mean_length": 78.25, "completions/min_length": 75.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.25, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.8964411020278931, "rewards/meter/std": 0.12688924372196198, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6750000715255737, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.6016985774040222, "rewards/total_composite/std": 0.1109333410859108, "reward": 0.6016985774040222, "reward_std": 0.11093335598707199, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02804330736398697, "sampling/sampling_logp_difference/max": 0.9042620658874512, "sampling/importance_sampling_ratio/min": 0.4048405587673187, "sampling/importance_sampling_ratio/mean": 1.0003811120986938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11404913989827037, "clip_ratio/low_mean": 0.012610982405021787, "clip_ratio/low_min": 0.012610982405021787, "clip_ratio/high_mean": 0.006666666595265269, "clip_ratio/high_max": 0.006666666595265269, "clip_ratio/region_mean": 0.019277649000287056, "reward_total_mean": 0.6016985774040222, "reward_meter_mean": 0.8964411020278931, "reward_meter_std": 0.12688924372196198, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6750000715255737, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.6016985774040222, "reward_total_composite_std": 0.1109333410859108, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1595.0} {"timestamp_utc": "2026-04-11T22:53:06Z", "mode": "train", "global_step": 763, "epoch": 0.03064626260191991, "loss": 0.0481, "grad_norm": 1.2447540760040283, "learning_rate": 7.690909090909091e-06, "num_tokens": 1684742.0, "completions/mean_length": 438.125, "completions/min_length": 391.0, "completions/max_length": 498.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 438.125, "completions/min_terminated_length": 391.0, "completions/max_terminated_length": 498.0, "rewards/meter/mean": 0.9878873825073242, "rewards/meter/std": 0.019713442772626877, "rewards/count_adherence/mean": 0.2142857164144516, "rewards/count_adherence/std": 0.07636035978794098, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5576170682907104, "rewards/repeat_penalty/std": 0.0946657732129097, "rewards/total_composite/mean": 0.12146620452404022, "rewards/total_composite/std": 0.05504889413714409, "reward": 0.12146620452404022, "reward_std": 0.05504889413714409, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010950884781777859, "sampling/sampling_logp_difference/max": 1.4895976781845093, "sampling/importance_sampling_ratio/min": 0.22546334564685822, "sampling/importance_sampling_ratio/mean": 1.0031057596206665, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07848789915442467, "clip_ratio/low_mean": 0.005670643062330782, "clip_ratio/low_min": 0.005670643062330782, "clip_ratio/high_mean": 0.005744358117226511, "clip_ratio/high_max": 0.005744358117226511, "clip_ratio/region_mean": 0.011415001179557294, "reward_total_mean": 0.12146620452404022, "reward_meter_mean": 0.9878873825073242, "reward_meter_std": 0.019713442772626877, "reward_count_adherence_mean": 0.2142857164144516, "reward_count_adherence_std": 0.07636035978794098, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5576170682907104, "reward_repeat_penalty_std": 0.0946657732129097, "reward_total_composite_mean": 0.12146620452404022, "reward_total_composite_std": 0.05504889413714409, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1596.0} {"timestamp_utc": "2026-04-11T22:53:11Z", "mode": "train", "global_step": 764, "epoch": 0.030686428083704864, "loss": 0.0189, "grad_norm": 5.640087127685547, "learning_rate": 7.687878787878788e-06, "num_tokens": 1686539.0, "completions/mean_length": 63.625, "completions/min_length": 60.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.874701976776123, "rewards/meter/std": 0.18508003652095795, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7850605845451355, "rewards/total_composite/std": 0.2699912190437317, "reward": 0.7850605845451355, "reward_std": 0.2699912190437317, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032904475927352905, "sampling/sampling_logp_difference/max": 1.8050861358642578, "sampling/importance_sampling_ratio/min": 0.16446030139923096, "sampling/importance_sampling_ratio/mean": 0.9924401044845581, "sampling/importance_sampling_ratio/max": 1.438651442527771, "entropy": 0.1180117940530181, "clip_ratio/low_mean": 0.005871212342754006, "clip_ratio/low_min": 0.005871212342754006, "clip_ratio/high_mean": 0.02714023506268859, "clip_ratio/high_max": 0.02714023506268859, "clip_ratio/region_mean": 0.033011447405442595, "reward_total_mean": 0.7850605845451355, "reward_meter_mean": 0.874701976776123, "reward_meter_std": 0.18508003652095795, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.7850605845451355, "reward_total_composite_std": 0.2699912190437317, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1597.0} {"timestamp_utc": "2026-04-11T22:53:16Z", "mode": "train", "global_step": 765, "epoch": 0.030726593565489818, "loss": 0.0008, "grad_norm": 4.968373775482178, "learning_rate": 7.684848484848485e-06, "num_tokens": 1688548.0, "completions/mean_length": 87.125, "completions/min_length": 87.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.125, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9950021505355835, "rewards/meter/std": 8.806584810372442e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6500000357627869, "rewards/repeat_penalty/std": 0.1414213478565216, "rewards/total_composite/mean": 0.6467622518539429, "rewards/total_composite/std": 0.14079822599887848, "reward": 0.6467622518539429, "reward_std": 0.14079821109771729, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004592791199684143, "sampling/sampling_logp_difference/max": 1.7399308681488037, "sampling/importance_sampling_ratio/min": 0.1755325347185135, "sampling/importance_sampling_ratio/mean": 0.9999979138374329, "sampling/importance_sampling_ratio/max": 1.767101764678955, "entropy": 0.008648848335724324, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0028409091755747795, "clip_ratio/high_max": 0.0028409091755747795, "clip_ratio/region_mean": 0.0028409091755747795, "reward_total_mean": 0.6467622518539429, "reward_meter_mean": 0.9950021505355835, "reward_meter_std": 8.806584810372442e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6500000357627869, "reward_repeat_penalty_std": 0.1414213478565216, "reward_total_composite_mean": 0.6467622518539429, "reward_total_composite_std": 0.14079822599887848, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1598.0} {"timestamp_utc": "2026-04-11T22:53:21Z", "mode": "train", "global_step": 766, "epoch": 0.030766759047274772, "loss": 0.0014, "grad_norm": 10.79392147064209, "learning_rate": 7.681818181818183e-06, "num_tokens": 1690236.0, "completions/mean_length": 52.0, "completions/min_length": 51.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8247180581092834, "rewards/meter/std": 0.3005145490169525, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8247180581092834, "rewards/total_composite/std": 0.3005145490169525, "reward": 0.8247180581092834, "reward_std": 0.3005145490169525, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04232970252633095, "sampling/sampling_logp_difference/max": 1.6089832782745361, "sampling/importance_sampling_ratio/min": 0.2000909447669983, "sampling/importance_sampling_ratio/mean": 0.997559666633606, "sampling/importance_sampling_ratio/max": 1.6643987894058228, "entropy": 0.19545143470168114, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/high_mean": 0.03126057400368154, "clip_ratio/high_max": 0.03126057400368154, "clip_ratio/region_mean": 0.03606826649047434, "reward_total_mean": 0.8247180581092834, "reward_meter_mean": 0.8247180581092834, "reward_meter_std": 0.3005145490169525, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8247180581092834, "reward_total_composite_std": 0.3005145490169525, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1599.0} {"timestamp_utc": "2026-04-11T22:53:26Z", "mode": "train", "global_step": 767, "epoch": 0.030806924529059726, "loss": 0.0103, "grad_norm": 12.172296524047852, "learning_rate": 7.678787878787878e-06, "num_tokens": 1692065.0, "completions/mean_length": 72.625, "completions/min_length": 67.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.3632194995880127, "rewards/meter/std": 0.32341283559799194, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.36088046431541443, "rewards/total_composite/std": 0.3263150453567505, "reward": 0.36088046431541443, "reward_std": 0.3263150453567505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07868395745754242, "sampling/sampling_logp_difference/max": 5.154880046844482, "sampling/importance_sampling_ratio/min": 0.005771172232925892, "sampling/importance_sampling_ratio/mean": 0.9972518086433411, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46072521805763245, "clip_ratio/low_mean": 0.03220835281535983, "clip_ratio/low_min": 0.03220835281535983, "clip_ratio/high_mean": 0.026226354064419866, "clip_ratio/high_max": 0.026226354064419866, "clip_ratio/region_mean": 0.058434706879779696, "reward_total_mean": 0.36088046431541443, "reward_meter_mean": 0.3632194995880127, "reward_meter_std": 0.32341283559799194, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.36088046431541443, "reward_total_composite_std": 0.3263150453567505, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1600.0} {"timestamp_utc": "2026-04-11T22:53:31Z", "mode": "train", "global_step": 768, "epoch": 0.03084709001084468, "loss": -0.0095, "grad_norm": 2.0709922313690186, "learning_rate": 7.675757575757577e-06, "num_tokens": 1694344.0, "completions/mean_length": 107.875, "completions/min_length": 104.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.875, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.7045435905456543, "rewards/meter/std": 0.13543730974197388, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5636348724365234, "rewards/total_composite/std": 0.10834985971450806, "reward": 0.5636348724365234, "reward_std": 0.10834983736276627, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010335095226764679, "sampling/sampling_logp_difference/max": 0.9596023559570312, "sampling/importance_sampling_ratio/min": 0.38304516673088074, "sampling/importance_sampling_ratio/mean": 1.0031026601791382, "sampling/importance_sampling_ratio/max": 1.4763460159301758, "entropy": 0.09003173373639584, "clip_ratio/low_mean": 0.0059091257862746716, "clip_ratio/low_min": 0.0059091257862746716, "clip_ratio/high_mean": 0.0022935778833925724, "clip_ratio/high_max": 0.0022935778833925724, "clip_ratio/region_mean": 0.008202703669667244, "reward_total_mean": 0.5636348724365234, "reward_meter_mean": 0.7045435905456543, "reward_meter_std": 0.13543730974197388, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5636348724365234, "reward_total_composite_std": 0.10834985971450806, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1601.0} {"timestamp_utc": "2026-04-11T22:53:37Z", "mode": "train", "global_step": 769, "epoch": 0.030887255492629634, "loss": 0.0714, "grad_norm": 7.707916736602783, "learning_rate": 7.672727272727273e-06, "num_tokens": 1696973.0, "completions/mean_length": 143.625, "completions/min_length": 126.0, "completions/max_length": 172.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.625, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 172.0, "rewards/meter/mean": 0.7441333532333374, "rewards/meter/std": 0.4554755687713623, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6174242496490479, "rewards/repeat_penalty/std": 0.1036364957690239, "rewards/total_composite/mean": 0.4105571210384369, "rewards/total_composite/std": 0.2803230583667755, "reward": 0.4105571210384369, "reward_std": 0.2803230583667755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03509281575679779, "sampling/sampling_logp_difference/max": 4.862173557281494, "sampling/importance_sampling_ratio/min": 0.0077336556278169155, "sampling/importance_sampling_ratio/mean": 0.9966462850570679, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0793102509342134, "clip_ratio/low_mean": 0.009834110038354993, "clip_ratio/low_min": 0.009834110038354993, "clip_ratio/high_mean": 0.0102224723668769, "clip_ratio/high_max": 0.0102224723668769, "clip_ratio/region_mean": 0.020056582405231893, "reward_total_mean": 0.4105571210384369, "reward_meter_mean": 0.7441333532333374, "reward_meter_std": 0.4554755687713623, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6174242496490479, "reward_repeat_penalty_std": 0.1036364957690239, "reward_total_composite_mean": 0.4105571210384369, "reward_total_composite_std": 0.2803230583667755, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1602.0} {"timestamp_utc": "2026-04-11T22:53:47Z", "mode": "train", "global_step": 770, "epoch": 0.030927420974414588, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.66969696969697e-06, "num_tokens": 1698965.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.8939934968948364, "rewards/meter/std": 0.13721102476119995, "rewards/count_adherence/mean": 0.7828947305679321, "rewards/count_adherence/std": 0.03373000770807266, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5402884483337402, "rewards/repeat_penalty/std": 0.2157072126865387, "rewards/total_composite/mean": 0.3744324743747711, "rewards/total_composite/std": 0.17441345751285553, "reward": 0.3744324743747711, "reward_std": 0.17441345751285553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.3744324743747711, "reward_meter_mean": 0.8939934968948364, "reward_meter_std": 0.13721102476119995, "reward_count_adherence_mean": 0.7828947305679321, "reward_count_adherence_std": 0.03373000770807266, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5402884483337402, "reward_repeat_penalty_std": 0.2157072126865387, "reward_total_composite_mean": 0.3744324743747711, "reward_total_composite_std": 0.17441345751285553, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1603.0} {"timestamp_utc": "2026-04-11T22:53:54Z", "mode": "train", "global_step": 771, "epoch": 0.030967586456199542, "loss": 0.0526, "grad_norm": 4.9908342361450195, "learning_rate": 7.666666666666667e-06, "num_tokens": 1702409.0, "completions/mean_length": 230.5, "completions/min_length": 180.0, "completions/max_length": 263.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 230.5, "completions/min_terminated_length": 180.0, "completions/max_terminated_length": 263.0, "rewards/meter/mean": 0.7852883338928223, "rewards/meter/std": 0.3382474482059479, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5688130855560303, "rewards/repeat_penalty/std": 0.20688475668430328, "rewards/total_composite/mean": 0.3491661250591278, "rewards/total_composite/std": 0.22600221633911133, "reward": 0.3491661250591278, "reward_std": 0.22600221633911133, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018919790163636208, "sampling/sampling_logp_difference/max": 1.5345475673675537, "sampling/importance_sampling_ratio/min": 0.21555320918560028, "sampling/importance_sampling_ratio/mean": 1.0030508041381836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13555688876658678, "clip_ratio/low_mean": 0.004294316866435111, "clip_ratio/low_min": 0.004294316866435111, "clip_ratio/high_mean": 0.010813763190526515, "clip_ratio/high_max": 0.010813763190526515, "clip_ratio/region_mean": 0.015108080056961626, "reward_total_mean": 0.3491661250591278, "reward_meter_mean": 0.7852883338928223, "reward_meter_std": 0.3382474482059479, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5688130855560303, "reward_repeat_penalty_std": 0.20688475668430328, "reward_total_composite_mean": 0.3491661250591278, "reward_total_composite_std": 0.22600221633911133, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1604.0} {"timestamp_utc": "2026-04-11T22:54:00Z", "mode": "train", "global_step": 772, "epoch": 0.031007751937984496, "loss": -0.1232, "grad_norm": 5.677700042724609, "learning_rate": 7.663636363636364e-06, "num_tokens": 1704516.0, "completions/mean_length": 99.375, "completions/min_length": 90.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.375, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9852504730224609, "rewards/meter/std": 0.008375532925128937, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7785714864730835, "rewards/repeat_penalty/std": 0.039677999913692474, "rewards/total_composite/mean": 0.6195021271705627, "rewards/total_composite/std": 0.05528084933757782, "reward": 0.6195021271705627, "reward_std": 0.05528085306286812, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020249057561159134, "sampling/sampling_logp_difference/max": 1.5644612312316895, "sampling/importance_sampling_ratio/min": 0.2092006951570511, "sampling/importance_sampling_ratio/mean": 1.0006357431411743, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08717024885118008, "clip_ratio/low_mean": 0.015005706925876439, "clip_ratio/low_min": 0.015005706925876439, "clip_ratio/high_mean": 0.006097560748457909, "clip_ratio/high_max": 0.006097560748457909, "clip_ratio/region_mean": 0.021103267674334347, "reward_total_mean": 0.6195021271705627, "reward_meter_mean": 0.9852504730224609, "reward_meter_std": 0.008375532925128937, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7785714864730835, "reward_repeat_penalty_std": 0.039677999913692474, "reward_total_composite_mean": 0.6195021271705627, "reward_total_composite_std": 0.05528084933757782, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1605.0} {"timestamp_utc": "2026-04-11T22:54:09Z", "mode": "train", "global_step": 773, "epoch": 0.03104791741976945, "loss": 0.0245, "grad_norm": 5.519010066986084, "learning_rate": 7.660606060606062e-06, "num_tokens": 1709201.0, "completions/mean_length": 409.625, "completions/min_length": 346.0, "completions/max_length": 460.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 409.625, "completions/min_terminated_length": 346.0, "completions/max_terminated_length": 460.0, "rewards/meter/mean": 0.9907213449478149, "rewards/meter/std": 0.006963523104786873, "rewards/count_adherence/mean": 0.3571428656578064, "rewards/count_adherence/std": 0.07636035233736038, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.45148104429244995, "rewards/repeat_penalty/std": 0.23929926753044128, "rewards/total_composite/mean": 0.17074038088321686, "rewards/total_composite/std": 0.10653632879257202, "reward": 0.17074038088321686, "reward_std": 0.10653632134199142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02135242149233818, "sampling/sampling_logp_difference/max": 5.103124141693115, "sampling/importance_sampling_ratio/min": 0.006077729165554047, "sampling/importance_sampling_ratio/mean": 0.9987974166870117, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07645870675332844, "clip_ratio/low_mean": 0.003349777136463672, "clip_ratio/low_min": 0.003349777136463672, "clip_ratio/high_mean": 0.008041649358347058, "clip_ratio/high_max": 0.008041649358347058, "clip_ratio/region_mean": 0.01139142649481073, "reward_total_mean": 0.17074038088321686, "reward_meter_mean": 0.9907213449478149, "reward_meter_std": 0.006963523104786873, "reward_count_adherence_mean": 0.3571428656578064, "reward_count_adherence_std": 0.07636035233736038, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.45148104429244995, "reward_repeat_penalty_std": 0.23929926753044128, "reward_total_composite_mean": 0.17074038088321686, "reward_total_composite_std": 0.10653632879257202, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1606.0} {"timestamp_utc": "2026-04-11T22:54:14Z", "mode": "train", "global_step": 774, "epoch": 0.031088082901554404, "loss": -0.0134, "grad_norm": 6.2788872718811035, "learning_rate": 7.657575757575757e-06, "num_tokens": 1711058.0, "completions/mean_length": 69.125, "completions/min_length": 65.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.7258920669555664, "rewards/meter/std": 0.40674278140068054, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7258920669555664, "rewards/total_composite/std": 0.40674278140068054, "reward": 0.7258920669555664, "reward_std": 0.40674278140068054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05154965817928314, "sampling/sampling_logp_difference/max": 2.2386436462402344, "sampling/importance_sampling_ratio/min": 0.1066029891371727, "sampling/importance_sampling_ratio/mean": 1.0082036256790161, "sampling/importance_sampling_ratio/max": 1.882003903388977, "entropy": 0.37627044692635536, "clip_ratio/low_mean": 0.005654420121572912, "clip_ratio/low_min": 0.005654420121572912, "clip_ratio/high_mean": 0.022313665016554296, "clip_ratio/high_max": 0.022313665016554296, "clip_ratio/region_mean": 0.027968085138127208, "reward_total_mean": 0.7258920669555664, "reward_meter_mean": 0.7258920669555664, "reward_meter_std": 0.40674278140068054, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7258920669555664, "reward_total_composite_std": 0.40674278140068054, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1607.0} {"timestamp_utc": "2026-04-11T22:54:20Z", "mode": "train", "global_step": 775, "epoch": 0.031128248383339358, "loss": 0.0636, "grad_norm": 2.807114601135254, "learning_rate": 7.654545454545456e-06, "num_tokens": 1713733.0, "completions/mean_length": 168.375, "completions/min_length": 147.0, "completions/max_length": 188.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.375, "completions/min_terminated_length": 147.0, "completions/max_terminated_length": 188.0, "rewards/meter/mean": 0.9743660688400269, "rewards/meter/std": 0.04519447684288025, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6060605645179749, "rewards/repeat_penalty/std": 0.13551926612854004, "rewards/total_composite/mean": 0.5430476665496826, "rewards/total_composite/std": 0.16461656987667084, "reward": 0.5430476665496826, "reward_std": 0.16461656987667084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026703810319304466, "sampling/sampling_logp_difference/max": 12.626859664916992, "sampling/importance_sampling_ratio/min": 3.2826496862981003e-06, "sampling/importance_sampling_ratio/mean": 0.9989210367202759, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06733021000400186, "clip_ratio/low_mean": 0.00418766331858933, "clip_ratio/low_min": 0.00418766331858933, "clip_ratio/high_mean": 0.007139897206798196, "clip_ratio/high_max": 0.007139897206798196, "clip_ratio/region_mean": 0.011327560525387526, "reward_total_mean": 0.5430476665496826, "reward_meter_mean": 0.9743660688400269, "reward_meter_std": 0.04519447684288025, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6060605645179749, "reward_repeat_penalty_std": 0.13551926612854004, "reward_total_composite_mean": 0.5430476665496826, "reward_total_composite_std": 0.16461656987667084, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1608.0} {"timestamp_utc": "2026-04-11T22:54:25Z", "mode": "train", "global_step": 776, "epoch": 0.03116841386512431, "loss": 0.0012, "grad_norm": 7.4227614402771, "learning_rate": 7.651515151515152e-06, "num_tokens": 1715409.0, "completions/mean_length": 54.5, "completions/min_length": 53.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9904471635818481, "rewards/meter/std": 0.002613567281514406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9904471635818481, "rewards/total_composite/std": 0.002613567281514406, "reward": 0.9904471635818481, "reward_std": 0.002613575430586934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024185476824641228, "sampling/sampling_logp_difference/max": 1.0824594497680664, "sampling/importance_sampling_ratio/min": 0.3387613296508789, "sampling/importance_sampling_ratio/mean": 0.9967302083969116, "sampling/importance_sampling_ratio/max": 1.4630638360977173, "entropy": 0.09682174911722541, "clip_ratio/low_mean": 0.009263938991352916, "clip_ratio/low_min": 0.009263938991352916, "clip_ratio/high_mean": 0.009050324792042375, "clip_ratio/high_max": 0.009050324792042375, "clip_ratio/region_mean": 0.01831426378339529, "reward_total_mean": 0.9904471635818481, "reward_meter_mean": 0.9904471635818481, "reward_meter_std": 0.002613567281514406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9904471635818481, "reward_total_composite_std": 0.002613567281514406, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1609.0} {"timestamp_utc": "2026-04-11T22:54:29Z", "mode": "train", "global_step": 777, "epoch": 0.031208579346909265, "loss": 0.0046, "grad_norm": 8.215063095092773, "learning_rate": 7.648484848484849e-06, "num_tokens": 1716925.0, "completions/mean_length": 36.5, "completions/min_length": 36.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.978127121925354, "rewards/meter/std": 0.0063257296569645405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.978127121925354, "rewards/total_composite/std": 0.0063257296569645405, "reward": 0.978127121925354, "reward_std": 0.006325736176222563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03158137574791908, "sampling/sampling_logp_difference/max": 1.4631445407867432, "sampling/importance_sampling_ratio/min": 0.23150713741779327, "sampling/importance_sampling_ratio/mean": 1.0036948919296265, "sampling/importance_sampling_ratio/max": 1.6868826150894165, "entropy": 0.11897748988121748, "clip_ratio/low_mean": 0.013795045437291265, "clip_ratio/low_min": 0.013795045437291265, "clip_ratio/high_mean": 0.013795045204460621, "clip_ratio/high_max": 0.013795045204460621, "clip_ratio/region_mean": 0.027590090641751885, "reward_total_mean": 0.978127121925354, "reward_meter_mean": 0.978127121925354, "reward_meter_std": 0.0063257296569645405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.978127121925354, "reward_total_composite_std": 0.0063257296569645405, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1610.0} {"timestamp_utc": "2026-04-11T22:54:34Z", "mode": "train", "global_step": 778, "epoch": 0.03124874482869422, "loss": 0.014, "grad_norm": 5.06115198135376, "learning_rate": 7.645454545454546e-06, "num_tokens": 1718951.0, "completions/mean_length": 72.25, "completions/min_length": 68.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9873093366622925, "rewards/meter/std": 0.01036460418254137, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9873093366622925, "rewards/total_composite/std": 0.01036460418254137, "reward": 0.9873093366622925, "reward_std": 0.010364595800638199, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03534568101167679, "sampling/sampling_logp_difference/max": 0.8904938697814941, "sampling/importance_sampling_ratio/min": 0.4104529917240143, "sampling/importance_sampling_ratio/mean": 1.0149582624435425, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25925541296601295, "clip_ratio/low_mean": 0.02250340231694281, "clip_ratio/low_min": 0.02250340231694281, "clip_ratio/high_mean": 0.020928236190229654, "clip_ratio/high_max": 0.020928236190229654, "clip_ratio/region_mean": 0.043431638507172465, "reward_total_mean": 0.9873093366622925, "reward_meter_mean": 0.9873093366622925, "reward_meter_std": 0.01036460418254137, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9873093366622925, "reward_total_composite_std": 0.01036460418254137, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1611.0} {"timestamp_utc": "2026-04-11T22:54:45Z", "mode": "train", "global_step": 779, "epoch": 0.03128891031047917, "loss": -0.1238, "grad_norm": 1.4244983196258545, "learning_rate": 7.642424242424244e-06, "num_tokens": 1720483.0, "completions/mean_length": 182.5, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 72.66667175292969, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.6190794110298157, "rewards/meter/std": 0.3674232065677643, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5616785287857056, "rewards/total_composite/std": 0.43257132172584534, "reward": 0.5616785287857056, "reward_std": 0.43257129192352295, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07225891947746277, "sampling/sampling_logp_difference/max": 1.2201738357543945, "sampling/importance_sampling_ratio/min": 0.2951788604259491, "sampling/importance_sampling_ratio/mean": 1.0154547691345215, "sampling/importance_sampling_ratio/max": 1.9660524129867554, "entropy": 0.38310878723859787, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/high_mean": 0.030813875840976834, "clip_ratio/high_max": 0.030813875840976834, "clip_ratio/region_mean": 0.039375519612804055, "reward_total_mean": 0.5616785287857056, "reward_meter_mean": 0.6190794110298157, "reward_meter_std": 0.3674232065677643, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5616785287857056, "reward_total_composite_std": 0.43257132172584534, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1612.0} {"timestamp_utc": "2026-04-11T22:54:56Z", "mode": "train", "global_step": 780, "epoch": 0.03132907579226413, "loss": -0.0498, "grad_norm": 3.119354009628296, "learning_rate": 7.639393939393939e-06, "num_tokens": 1722862.0, "completions/mean_length": 174.375, "completions/min_length": 111.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 126.14286041259766, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.6168656349182129, "rewards/meter/std": 0.4164867103099823, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.7492559552192688, "rewards/repeat_penalty/std": 0.14768067002296448, "rewards/total_composite/mean": 0.4575069546699524, "rewards/total_composite/std": 0.3478102684020996, "reward": 0.4575069546699524, "reward_std": 0.3478102684020996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03274521231651306, "sampling/sampling_logp_difference/max": 1.078242540359497, "sampling/importance_sampling_ratio/min": 0.34019285440444946, "sampling/importance_sampling_ratio/mean": 1.0024807453155518, "sampling/importance_sampling_ratio/max": 1.7341415882110596, "entropy": 0.24183030799031258, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/high_mean": 0.029140884289518, "clip_ratio/high_max": 0.029140884289518, "clip_ratio/region_mean": 0.0339485767763108, "reward_total_mean": 0.4575069546699524, "reward_meter_mean": 0.6168656349182129, "reward_meter_std": 0.4164867103099823, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.7492559552192688, "reward_repeat_penalty_std": 0.14768067002296448, "reward_total_composite_mean": 0.4575069546699524, "reward_total_composite_std": 0.3478102684020996, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1613.0} {"timestamp_utc": "2026-04-11T22:55:01Z", "mode": "train", "global_step": 781, "epoch": 0.03136924127404908, "loss": -0.0116, "grad_norm": 12.43602180480957, "learning_rate": 7.636363636363638e-06, "num_tokens": 1724537.0, "completions/mean_length": 47.375, "completions/min_length": 44.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.5619980096817017, "rewards/meter/std": 0.3257886469364166, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.4856577217578888, "rewards/total_composite/std": 0.3285104036331177, "reward": 0.4856577217578888, "reward_std": 0.3285104036331177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09212920814752579, "sampling/sampling_logp_difference/max": 1.6957062482833862, "sampling/importance_sampling_ratio/min": 0.18346960842609406, "sampling/importance_sampling_ratio/mean": 0.9970747232437134, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9118200056254864, "clip_ratio/low_mean": 0.04398810095153749, "clip_ratio/low_min": 0.04398810095153749, "clip_ratio/high_mean": 0.016590908635407686, "clip_ratio/high_max": 0.016590908635407686, "clip_ratio/region_mean": 0.060579009586945176, "reward_total_mean": 0.4856577217578888, "reward_meter_mean": 0.5619980096817017, "reward_meter_std": 0.3257886469364166, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.4856577217578888, "reward_total_composite_std": 0.3285104036331177, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1614.0} {"timestamp_utc": "2026-04-11T22:55:10Z", "mode": "train", "global_step": 782, "epoch": 0.031409406755834035, "loss": -0.1573, "grad_norm": 1.6532626152038574, "learning_rate": 7.633333333333334e-06, "num_tokens": 1726623.0, "completions/mean_length": 166.75, "completions/min_length": 111.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 117.42857360839844, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.6895759105682373, "rewards/meter/std": 0.26801931858062744, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.824999988079071, "rewards/repeat_penalty/std": 0.19820624589920044, "rewards/total_composite/mean": 0.5303109288215637, "rewards/total_composite/std": 0.28132033348083496, "reward": 0.5303109288215637, "reward_std": 0.28132033348083496, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03518048673868179, "sampling/sampling_logp_difference/max": 1.1620054244995117, "sampling/importance_sampling_ratio/min": 0.31285813450813293, "sampling/importance_sampling_ratio/mean": 1.006204605102539, "sampling/importance_sampling_ratio/max": 1.9478799104690552, "entropy": 0.2663711039349437, "clip_ratio/low_mean": 0.008460594224743545, "clip_ratio/low_min": 0.008460594224743545, "clip_ratio/high_mean": 0.013982121949084103, "clip_ratio/high_max": 0.013982121949084103, "clip_ratio/region_mean": 0.022442716173827648, "reward_total_mean": 0.5303109288215637, "reward_meter_mean": 0.6895759105682373, "reward_meter_std": 0.26801931858062744, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.824999988079071, "reward_repeat_penalty_std": 0.19820624589920044, "reward_total_composite_mean": 0.5303109288215637, "reward_total_composite_std": 0.28132033348083496, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1615.0} {"timestamp_utc": "2026-04-11T22:55:15Z", "mode": "train", "global_step": 783, "epoch": 0.03144957223761899, "loss": 0.0077, "grad_norm": 4.994919776916504, "learning_rate": 7.630303030303031e-06, "num_tokens": 1728611.0, "completions/mean_length": 79.5, "completions/min_length": 76.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.5, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.6204426288604736, "rewards/meter/std": 0.3357177972793579, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6204426288604736, "rewards/total_composite/std": 0.3357177972793579, "reward": 0.6204426288604736, "reward_std": 0.3357177972793579, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04477819800376892, "sampling/sampling_logp_difference/max": 1.0638208389282227, "sampling/importance_sampling_ratio/min": 0.34513458609580994, "sampling/importance_sampling_ratio/mean": 1.011016845703125, "sampling/importance_sampling_ratio/max": 1.7397865056991577, "entropy": 0.3720816671848297, "clip_ratio/low_mean": 0.017455301131121814, "clip_ratio/low_min": 0.017455301131121814, "clip_ratio/high_mean": 0.013986280770041049, "clip_ratio/high_max": 0.013986280770041049, "clip_ratio/region_mean": 0.03144158190116286, "reward_total_mean": 0.6204426288604736, "reward_meter_mean": 0.6204426288604736, "reward_meter_std": 0.3357177972793579, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6204426288604736, "reward_total_composite_std": 0.3357177972793579, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1616.0} {"timestamp_utc": "2026-04-11T22:55:20Z", "mode": "train", "global_step": 784, "epoch": 0.03148973771940394, "loss": 0.0052, "grad_norm": 7.485208988189697, "learning_rate": 7.627272727272727e-06, "num_tokens": 1730350.0, "completions/mean_length": 61.375, "completions/min_length": 56.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.5055776834487915, "rewards/meter/std": 0.32300832867622375, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.453016996383667, "rewards/total_composite/std": 0.33306846022605896, "reward": 0.453016996383667, "reward_std": 0.33306846022605896, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.047304026782512665, "sampling/sampling_logp_difference/max": 1.7305419445037842, "sampling/importance_sampling_ratio/min": 0.17718835175037384, "sampling/importance_sampling_ratio/mean": 0.9948754906654358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2184693105518818, "clip_ratio/low_mean": 0.018428327050060034, "clip_ratio/low_min": 0.018428327050060034, "clip_ratio/high_mean": 0.012099922401830554, "clip_ratio/high_max": 0.012099922401830554, "clip_ratio/region_mean": 0.030528249451890588, "reward_total_mean": 0.453016996383667, "reward_meter_mean": 0.5055776834487915, "reward_meter_std": 0.32300832867622375, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.453016996383667, "reward_total_composite_std": 0.33306846022605896, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1617.0} {"timestamp_utc": "2026-04-11T22:55:25Z", "mode": "train", "global_step": 785, "epoch": 0.0315299032011889, "loss": 0.0036, "grad_norm": 6.059953212738037, "learning_rate": 7.6242424242424254e-06, "num_tokens": 1732524.0, "completions/mean_length": 104.75, "completions/min_length": 99.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.75, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.49319469928741455, "rewards/meter/std": 0.2711309790611267, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.46458256244659424, "rewards/total_composite/std": 0.2836271822452545, "reward": 0.46458256244659424, "reward_std": 0.2836271822452545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042036473751068115, "sampling/sampling_logp_difference/max": 1.2848865985870361, "sampling/importance_sampling_ratio/min": 0.27668195962905884, "sampling/importance_sampling_ratio/mean": 1.0086406469345093, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31474705785512924, "clip_ratio/low_mean": 0.010869022691622376, "clip_ratio/low_min": 0.010869022691622376, "clip_ratio/high_mean": 0.01767290150746703, "clip_ratio/high_max": 0.01767290150746703, "clip_ratio/region_mean": 0.028541924199089408, "reward_total_mean": 0.46458256244659424, "reward_meter_mean": 0.49319469928741455, "reward_meter_std": 0.2711309790611267, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.46458256244659424, "reward_total_composite_std": 0.2836271822452545, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1618.0} {"timestamp_utc": "2026-04-11T22:55:34Z", "mode": "train", "global_step": 786, "epoch": 0.03157006868297385, "loss": -0.0567, "grad_norm": 2.024092435836792, "learning_rate": 7.621212121212122e-06, "num_tokens": 1737190.0, "completions/mean_length": 368.25, "completions/min_length": 322.0, "completions/max_length": 406.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 368.25, "completions/min_terminated_length": 322.0, "completions/max_terminated_length": 406.0, "rewards/meter/mean": 0.7014279365539551, "rewards/meter/std": 0.43751856684684753, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5806276798248291, "rewards/repeat_penalty/std": 0.06668182462453842, "rewards/total_composite/mean": 0.3579794764518738, "rewards/total_composite/std": 0.22696760296821594, "reward": 0.3579794764518738, "reward_std": 0.22696760296821594, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012620110996067524, "sampling/sampling_logp_difference/max": 1.491776466369629, "sampling/importance_sampling_ratio/min": 0.2249726504087448, "sampling/importance_sampling_ratio/mean": 1.0009483098983765, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07080868305638433, "clip_ratio/low_mean": 0.004607091657817364, "clip_ratio/low_min": 0.004607091657817364, "clip_ratio/high_mean": 0.0042421949619892985, "clip_ratio/high_max": 0.0042421949619892985, "clip_ratio/region_mean": 0.008849286619806662, "reward_total_mean": 0.3579794764518738, "reward_meter_mean": 0.7014279365539551, "reward_meter_std": 0.43751856684684753, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5806276798248291, "reward_repeat_penalty_std": 0.06668182462453842, "reward_total_composite_mean": 0.3579794764518738, "reward_total_composite_std": 0.22696760296821594, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1619.0} {"timestamp_utc": "2026-04-11T22:55:40Z", "mode": "train", "global_step": 787, "epoch": 0.031610234164758805, "loss": 0.0123, "grad_norm": 3.6839067935943604, "learning_rate": 7.618181818181819e-06, "num_tokens": 1739852.0, "completions/mean_length": 153.75, "completions/min_length": 135.0, "completions/max_length": 162.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 153.75, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 162.0, "rewards/meter/mean": 0.7929247617721558, "rewards/meter/std": 0.10911522805690765, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6818181872367859, "rewards/repeat_penalty/std": 0.06872082501649857, "rewards/total_composite/mean": 0.435272216796875, "rewards/total_composite/std": 0.09194310009479523, "reward": 0.435272216796875, "reward_std": 0.09194309264421463, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028897186741232872, "sampling/sampling_logp_difference/max": 1.9267759323120117, "sampling/importance_sampling_ratio/min": 0.14561693370342255, "sampling/importance_sampling_ratio/mean": 0.9994563460350037, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11316746287047863, "clip_ratio/low_mean": 0.01558087719604373, "clip_ratio/low_min": 0.01558087719604373, "clip_ratio/high_mean": 0.007280809339135885, "clip_ratio/high_max": 0.007280809339135885, "clip_ratio/region_mean": 0.022861686535179615, "reward_total_mean": 0.435272216796875, "reward_meter_mean": 0.7929247617721558, "reward_meter_std": 0.10911522805690765, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6818181872367859, "reward_repeat_penalty_std": 0.06872082501649857, "reward_total_composite_mean": 0.435272216796875, "reward_total_composite_std": 0.09194310009479523, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1620.0} {"timestamp_utc": "2026-04-11T22:55:44Z", "mode": "train", "global_step": 788, "epoch": 0.03165039964654376, "loss": -0.0051, "grad_norm": 7.856208324432373, "learning_rate": 7.6151515151515155e-06, "num_tokens": 1741588.0, "completions/mean_length": 58.0, "completions/min_length": 57.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.6469426155090332, "rewards/meter/std": 0.24868278205394745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6469426155090332, "rewards/total_composite/std": 0.24868278205394745, "reward": 0.6469426155090332, "reward_std": 0.24868276715278625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022744039073586464, "sampling/sampling_logp_difference/max": 0.6672461032867432, "sampling/importance_sampling_ratio/min": 0.5131197571754456, "sampling/importance_sampling_ratio/mean": 1.0034902095794678, "sampling/importance_sampling_ratio/max": 1.8282591104507446, "entropy": 0.16343743726611137, "clip_ratio/low_mean": 0.012860541231930256, "clip_ratio/low_min": 0.012860541231930256, "clip_ratio/high_mean": 0.01074189692735672, "clip_ratio/high_max": 0.01074189692735672, "clip_ratio/region_mean": 0.023602438159286976, "reward_total_mean": 0.6469426155090332, "reward_meter_mean": 0.6469426155090332, "reward_meter_std": 0.24868278205394745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6469426155090332, "reward_total_composite_std": 0.24868278205394745, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1621.0} {"timestamp_utc": "2026-04-11T22:55:49Z", "mode": "train", "global_step": 789, "epoch": 0.03169056512832871, "loss": 0.0131, "grad_norm": 13.922452926635742, "learning_rate": 7.612121212121213e-06, "num_tokens": 1743054.0, "completions/mean_length": 31.25, "completions/min_length": 30.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9813005924224854, "rewards/meter/std": 0.016189413145184517, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9813005924224854, "rewards/total_composite/std": 0.016189413145184517, "reward": 0.9813005924224854, "reward_std": 0.016189415007829666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033597491681575775, "sampling/sampling_logp_difference/max": 1.0770900249481201, "sampling/importance_sampling_ratio/min": 0.3405851721763611, "sampling/importance_sampling_ratio/mean": 0.9976152181625366, "sampling/importance_sampling_ratio/max": 1.8372166156768799, "entropy": 0.17746192403137684, "clip_ratio/low_mean": 0.008072916883975267, "clip_ratio/low_min": 0.008072916883975267, "clip_ratio/high_mean": 0.008072916883975267, "clip_ratio/high_max": 0.008072916883975267, "clip_ratio/region_mean": 0.016145833767950535, "reward_total_mean": 0.9813005924224854, "reward_meter_mean": 0.9813005924224854, "reward_meter_std": 0.016189413145184517, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9813005924224854, "reward_total_composite_std": 0.016189413145184517, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1622.0} {"timestamp_utc": "2026-04-11T22:55:54Z", "mode": "train", "global_step": 790, "epoch": 0.03173073061011367, "loss": -0.0039, "grad_norm": 6.977390766143799, "learning_rate": 7.609090909090909e-06, "num_tokens": 1745187.0, "completions/mean_length": 89.625, "completions/min_length": 85.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.625, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.8486406207084656, "rewards/meter/std": 0.2562169134616852, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6789125204086304, "rewards/total_composite/std": 0.20497353374958038, "reward": 0.6789125204086304, "reward_std": 0.20497353374958038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021279064938426018, "sampling/sampling_logp_difference/max": 0.814723014831543, "sampling/importance_sampling_ratio/min": 0.44276192784309387, "sampling/importance_sampling_ratio/mean": 1.0061084032058716, "sampling/importance_sampling_ratio/max": 1.8353419303894043, "entropy": 0.11805877834558487, "clip_ratio/low_mean": 0.009868005756288767, "clip_ratio/low_min": 0.009868005756288767, "clip_ratio/high_mean": 0.019598963437601924, "clip_ratio/high_max": 0.019598963437601924, "clip_ratio/region_mean": 0.02946696919389069, "reward_total_mean": 0.6789125204086304, "reward_meter_mean": 0.8486406207084656, "reward_meter_std": 0.2562169134616852, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6789125204086304, "reward_total_composite_std": 0.20497353374958038, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1623.0} {"timestamp_utc": "2026-04-11T22:56:01Z", "mode": "train", "global_step": 791, "epoch": 0.03177089609189862, "loss": -0.0395, "grad_norm": 1.8626375198364258, "learning_rate": 7.606060606060606e-06, "num_tokens": 1748773.0, "completions/mean_length": 247.25, "completions/min_length": 228.0, "completions/max_length": 258.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 247.25, "completions/min_terminated_length": 228.0, "completions/max_terminated_length": 258.0, "rewards/meter/mean": 0.9959322214126587, "rewards/meter/std": 0.004204806871712208, "rewards/count_adherence/mean": 0.910714328289032, "rewards/count_adherence/std": 0.07393559068441391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.49047619104385376, "rewards/repeat_penalty/std": 0.17105023562908173, "rewards/total_composite/mean": 0.4338645935058594, "rewards/total_composite/std": 0.12500374019145966, "reward": 0.4338645935058594, "reward_std": 0.12500372529029846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014479896053671837, "sampling/sampling_logp_difference/max": 2.833479881286621, "sampling/importance_sampling_ratio/min": 0.0588078573346138, "sampling/importance_sampling_ratio/mean": 1.0004253387451172, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.042171002831310034, "clip_ratio/low_mean": 0.0032260402804240584, "clip_ratio/low_min": 0.0032260402804240584, "clip_ratio/high_mean": 0.00779381615575403, "clip_ratio/high_max": 0.00779381615575403, "clip_ratio/region_mean": 0.011019856436178088, "reward_total_mean": 0.4338645935058594, "reward_meter_mean": 0.9959322214126587, "reward_meter_std": 0.004204806871712208, "reward_count_adherence_mean": 0.910714328289032, "reward_count_adherence_std": 0.07393559068441391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.49047619104385376, "reward_repeat_penalty_std": 0.17105023562908173, "reward_total_composite_mean": 0.4338645935058594, "reward_total_composite_std": 0.12500374019145966, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1624.0} {"timestamp_utc": "2026-04-11T22:56:06Z", "mode": "train", "global_step": 792, "epoch": 0.031811061573683574, "loss": 0.023, "grad_norm": 3.4738333225250244, "learning_rate": 7.603030303030303e-06, "num_tokens": 1751485.0, "completions/mean_length": 135.0, "completions/min_length": 124.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.0, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.991753101348877, "rewards/meter/std": 0.002629074966534972, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6611686944961548, "rewards/total_composite/std": 0.0017527303425595164, "reward": 0.6611686944961548, "reward_std": 0.0017527244053781033, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019971484318375587, "sampling/sampling_logp_difference/max": 2.0009312629699707, "sampling/importance_sampling_ratio/min": 0.13520930707454681, "sampling/importance_sampling_ratio/mean": 1.0068343877792358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0634374669753015, "clip_ratio/low_mean": 0.010955466306768358, "clip_ratio/low_min": 0.010955466306768358, "clip_ratio/high_mean": 0.00658258656039834, "clip_ratio/high_max": 0.00658258656039834, "clip_ratio/region_mean": 0.017538052867166698, "reward_total_mean": 0.6611686944961548, "reward_meter_mean": 0.991753101348877, "reward_meter_std": 0.002629074966534972, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6611686944961548, "reward_total_composite_std": 0.0017527303425595164, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1625.0} {"timestamp_utc": "2026-04-11T22:56:11Z", "mode": "train", "global_step": 793, "epoch": 0.03185122705546853, "loss": 0.0066, "grad_norm": 5.889811038970947, "learning_rate": 7.600000000000001e-06, "num_tokens": 1753253.0, "completions/mean_length": 56.0, "completions/min_length": 51.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9871419668197632, "rewards/meter/std": 0.005153529345989227, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.9050641655921936, "rewards/total_composite/std": 0.15341757237911224, "reward": 0.9050641655921936, "reward_std": 0.15341757237911224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.054076679050922394, "sampling/sampling_logp_difference/max": 2.979457139968872, "sampling/importance_sampling_ratio/min": 0.05082041397690773, "sampling/importance_sampling_ratio/mean": 0.9988625645637512, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15502143744379282, "clip_ratio/low_mean": 0.006658692145720124, "clip_ratio/low_min": 0.006658692145720124, "clip_ratio/high_mean": 0.0335467669647187, "clip_ratio/high_max": 0.0335467669647187, "clip_ratio/region_mean": 0.040205459110438824, "reward_total_mean": 0.9050641655921936, "reward_meter_mean": 0.9871419668197632, "reward_meter_std": 0.005153529345989227, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.9050641655921936, "reward_total_composite_std": 0.15341757237911224, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1626.0} {"timestamp_utc": "2026-04-11T22:56:16Z", "mode": "train", "global_step": 794, "epoch": 0.03189139253725348, "loss": 0.002, "grad_norm": 6.565411567687988, "learning_rate": 7.596969696969697e-06, "num_tokens": 1755042.0, "completions/mean_length": 71.625, "completions/min_length": 69.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.625, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9123142957687378, "rewards/meter/std": 0.09913700073957443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9123142957687378, "rewards/total_composite/std": 0.09913700073957443, "reward": 0.9123142957687378, "reward_std": 0.09913701564073563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034922052174806595, "sampling/sampling_logp_difference/max": 1.3885717391967773, "sampling/importance_sampling_ratio/min": 0.249431312084198, "sampling/importance_sampling_ratio/mean": 1.0043187141418457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15924948174506426, "clip_ratio/low_mean": 0.019052310031838715, "clip_ratio/low_min": 0.019052310031838715, "clip_ratio/high_mean": 0.007044379832223058, "clip_ratio/high_max": 0.007044379832223058, "clip_ratio/region_mean": 0.026096689864061773, "reward_total_mean": 0.9123142957687378, "reward_meter_mean": 0.9123142957687378, "reward_meter_std": 0.09913700073957443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9123142957687378, "reward_total_composite_std": 0.09913700073957443, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1627.0} {"timestamp_utc": "2026-04-11T22:56:21Z", "mode": "train", "global_step": 795, "epoch": 0.031931558019038436, "loss": 0.0046, "grad_norm": 2.6046133041381836, "learning_rate": 7.593939393939395e-06, "num_tokens": 1756866.0, "completions/mean_length": 73.0, "completions/min_length": 73.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.613559365272522, "rewards/meter/std": 0.046057041734457016, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.613559365272522, "rewards/total_composite/std": 0.046057041734457016, "reward": 0.613559365272522, "reward_std": 0.04605703800916672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01630816049873829, "sampling/sampling_logp_difference/max": 0.9375072717666626, "sampling/importance_sampling_ratio/min": 0.5393000841140747, "sampling/importance_sampling_ratio/mean": 1.0046606063842773, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07319271843880415, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/high_mean": 0.0034246575087308884, "clip_ratio/high_max": 0.0034246575087308884, "clip_ratio/region_mean": 0.01198630128055811, "reward_total_mean": 0.613559365272522, "reward_meter_mean": 0.613559365272522, "reward_meter_std": 0.046057041734457016, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.613559365272522, "reward_total_composite_std": 0.046057041734457016, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1628.0} {"timestamp_utc": "2026-04-11T22:56:28Z", "mode": "train", "global_step": 796, "epoch": 0.03197172350082339, "loss": 0.0182, "grad_norm": 1.5290881395339966, "learning_rate": 7.590909090909091e-06, "num_tokens": 1760371.0, "completions/mean_length": 250.125, "completions/min_length": 248.0, "completions/max_length": 264.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 250.125, "completions/min_terminated_length": 248.0, "completions/max_terminated_length": 264.0, "rewards/meter/mean": 0.6284428834915161, "rewards/meter/std": 0.1288241744041443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6098901033401489, "rewards/repeat_penalty/std": 0.015540807507932186, "rewards/total_composite/mean": 0.3849101662635803, "rewards/total_composite/std": 0.08409524708986282, "reward": 0.3849101662635803, "reward_std": 0.08409524708986282, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006375753786414862, "sampling/sampling_logp_difference/max": 1.0706424713134766, "sampling/importance_sampling_ratio/min": 0.3427882194519043, "sampling/importance_sampling_ratio/mean": 1.0002148151397705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.023278776556253433, "clip_ratio/low_mean": 0.0004734848625957966, "clip_ratio/low_min": 0.0004734848625957966, "clip_ratio/high_mean": 0.002516112755984068, "clip_ratio/high_max": 0.002516112755984068, "clip_ratio/region_mean": 0.0029895976185798645, "reward_total_mean": 0.3849101662635803, "reward_meter_mean": 0.6284428834915161, "reward_meter_std": 0.1288241744041443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6098901033401489, "reward_repeat_penalty_std": 0.015540807507932186, "reward_total_composite_mean": 0.3849101662635803, "reward_total_composite_std": 0.08409524708986282, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1629.0} {"timestamp_utc": "2026-04-11T22:56:33Z", "mode": "train", "global_step": 797, "epoch": 0.032011888982608344, "loss": -0.0121, "grad_norm": 8.541836738586426, "learning_rate": 7.587878787878788e-06, "num_tokens": 1762455.0, "completions/mean_length": 77.5, "completions/min_length": 72.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9952297210693359, "rewards/meter/std": 0.002311403863132, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.891382098197937, "rewards/total_composite/std": 0.1964196413755417, "reward": 0.891382098197937, "reward_std": 0.1964196413755417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034574542194604874, "sampling/sampling_logp_difference/max": 3.3448140621185303, "sampling/importance_sampling_ratio/min": 0.03526677191257477, "sampling/importance_sampling_ratio/mean": 0.9954510927200317, "sampling/importance_sampling_ratio/max": 1.482370376586914, "entropy": 0.14471599273383617, "clip_ratio/low_mean": 0.003246753243729472, "clip_ratio/low_min": 0.003246753243729472, "clip_ratio/high_mean": 0.009473072132095695, "clip_ratio/high_max": 0.009473072132095695, "clip_ratio/region_mean": 0.012719825375825167, "reward_total_mean": 0.891382098197937, "reward_meter_mean": 0.9952297210693359, "reward_meter_std": 0.002311403863132, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.891382098197937, "reward_total_composite_std": 0.1964196413755417, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1630.0} {"timestamp_utc": "2026-04-11T22:56:38Z", "mode": "train", "global_step": 798, "epoch": 0.0320520544643933, "loss": 0.0167, "grad_norm": 6.7140374183654785, "learning_rate": 7.584848484848486e-06, "num_tokens": 1764291.0, "completions/mean_length": 66.5, "completions/min_length": 64.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6813795566558838, "rewards/meter/std": 0.04148668423295021, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6813795566558838, "rewards/total_composite/std": 0.04148668423295021, "reward": 0.6813795566558838, "reward_std": 0.04148669168353081, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023791270330548286, "sampling/sampling_logp_difference/max": 0.9566974639892578, "sampling/importance_sampling_ratio/min": 0.5285674333572388, "sampling/importance_sampling_ratio/mean": 1.010377049446106, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1323620891198516, "clip_ratio/low_mean": 0.020607191254384816, "clip_ratio/low_min": 0.020607191254384816, "clip_ratio/high_mean": 0.00390625, "clip_ratio/high_max": 0.00390625, "clip_ratio/region_mean": 0.024513441254384816, "reward_total_mean": 0.6813795566558838, "reward_meter_mean": 0.6813795566558838, "reward_meter_std": 0.04148668423295021, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6813795566558838, "reward_total_composite_std": 0.04148668423295021, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1631.0} {"timestamp_utc": "2026-04-11T22:56:45Z", "mode": "train", "global_step": 799, "epoch": 0.03209221994617825, "loss": 0.0044, "grad_norm": 1.2270196676254272, "learning_rate": 7.581818181818183e-06, "num_tokens": 1767790.0, "completions/mean_length": 239.375, "completions/min_length": 237.0, "completions/max_length": 240.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 239.375, "completions/min_terminated_length": 237.0, "completions/max_terminated_length": 240.0, "rewards/meter/mean": 0.7462765574455261, "rewards/meter/std": 0.09833642095327377, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6000000238418579, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.38379937410354614, "rewards/total_composite/std": 0.05057300627231598, "reward": 0.38379937410354614, "reward_std": 0.050573013722896576, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007426128257066011, "sampling/sampling_logp_difference/max": 2.6020090579986572, "sampling/importance_sampling_ratio/min": 0.07412450760602951, "sampling/importance_sampling_ratio/mean": 0.9996190071105957, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.030161422211676836, "clip_ratio/low_mean": 0.003645833523478359, "clip_ratio/low_min": 0.003645833523478359, "clip_ratio/high_mean": 0.002109704539179802, "clip_ratio/high_max": 0.002109704539179802, "clip_ratio/region_mean": 0.005755538062658161, "reward_total_mean": 0.38379937410354614, "reward_meter_mean": 0.7462765574455261, "reward_meter_std": 0.09833642095327377, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6000000238418579, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.38379937410354614, "reward_total_composite_std": 0.05057300627231598, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1632.0} {"timestamp_utc": "2026-04-11T22:56:50Z", "mode": "train", "global_step": 800, "epoch": 0.032132385427963206, "loss": 0.0432, "grad_norm": 4.79632568359375, "learning_rate": 7.57878787878788e-06, "num_tokens": 1770025.0, "completions/mean_length": 119.375, "completions/min_length": 110.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.375, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.5170607566833496, "rewards/meter/std": 0.48065322637557983, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.4120614528656006, "rewards/total_composite/std": 0.386055052280426, "reward": 0.4120614528656006, "reward_std": 0.386055052280426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04103344678878784, "sampling/sampling_logp_difference/max": 1.1491317749023438, "sampling/importance_sampling_ratio/min": 0.3169117867946625, "sampling/importance_sampling_ratio/mean": 0.9977750182151794, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23464529775083065, "clip_ratio/low_mean": 0.024260351667180657, "clip_ratio/low_min": 0.024260351667180657, "clip_ratio/high_mean": 0.0143592240056023, "clip_ratio/high_max": 0.0143592240056023, "clip_ratio/region_mean": 0.03861957567278296, "reward_total_mean": 0.4120614528656006, "reward_meter_mean": 0.5170607566833496, "reward_meter_std": 0.48065322637557983, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.4120614528656006, "reward_total_composite_std": 0.386055052280426, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1633.0} {"timestamp_utc": "2026-04-11T22:58:16Z", "mode": "eval", "global_step": 800, "epoch": 0.032132385427963206, "eval_loss": NaN, "eval_runtime": 85.9157, "eval_samples_per_second": 1.21, "eval_steps_per_second": 0.151, "eval_num_tokens": 1770025.0, "eval_completions/mean_length": 234.29807692307693, "eval_completions/min_length": 63.84615384615385, "eval_completions/max_length": 461.9230769230769, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/mean_terminated_length": 216.9835216815655, "eval_completions/min_terminated_length": 63.84615384615385, "eval_completions/max_terminated_length": 414.46153846153845, "eval_rewards/meter/mean": 0.6419616112342248, "eval_rewards/meter/std": 0.3754527878302794, "eval_rewards/count_adherence/mean": 0.9552615697567279, "eval_rewards/count_adherence/std": 0.08013723160211857, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.6556147245260385, "eval_rewards/repeat_penalty/std": 0.24277657327743676, "eval_rewards/total_composite/mean": 0.3991968219096844, "eval_rewards/total_composite/std": 0.3098094039238416, "eval_reward": 0.3991968219096844, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0054089168552309275, "eval_sampling/sampling_logp_difference/max": 0.7075341939926147, "eval_sampling/importance_sampling_ratio/min": 0.5230464408030877, "eval_sampling/importance_sampling_ratio/mean": 1.0009051194557776, "eval_sampling/importance_sampling_ratio/max": 1.2961730773632343, "eval_entropy": 0.047992275741237864, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.3991968219096844, "eval_reward_meter_mean": 0.6419616112342248, "eval_reward_meter_std": 0.3754527878302794, "eval_reward_count_adherence_mean": 0.9552615697567279, "eval_reward_count_adherence_std": 0.08013723160211857, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.6556147245260385, "eval_reward_repeat_penalty_std": 0.24277657327743676, "eval_reward_total_composite_mean": 0.3991968219096844, "eval_reward_total_composite_std": 0.3098094039238416, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1633.0} {"timestamp_utc": "2026-04-11T22:58:23Z", "mode": "train", "global_step": 801, "epoch": 0.03217255090974816, "loss": 0.0018, "grad_norm": 4.925754070281982, "learning_rate": 7.5757575757575764e-06, "num_tokens": 1771810.0, "completions/mean_length": 67.125, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7430859804153442, "rewards/meter/std": 0.04014641046524048, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7430859804153442, "rewards/total_composite/std": 0.04014641046524048, "reward": 0.7430859804153442, "reward_std": 0.040146395564079285, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02094501443207264, "sampling/sampling_logp_difference/max": 1.098017692565918, "sampling/importance_sampling_ratio/min": 0.33353158831596375, "sampling/importance_sampling_ratio/mean": 1.0070921182632446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08063888642936945, "clip_ratio/low_mean": 0.018601843621581793, "clip_ratio/low_min": 0.018601843621581793, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.018601843621581793, "reward_total_mean": 0.7430859804153442, "reward_meter_mean": 0.7430859804153442, "reward_meter_std": 0.04014641046524048, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7430859804153442, "reward_total_composite_std": 0.04014641046524048, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1634.0} {"timestamp_utc": "2026-04-11T22:58:28Z", "mode": "train", "global_step": 802, "epoch": 0.032212716391533114, "loss": 0.0172, "grad_norm": 7.339356422424316, "learning_rate": 7.572727272727274e-06, "num_tokens": 1773359.0, "completions/mean_length": 37.625, "completions/min_length": 36.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.13143374025821686, "rewards/meter/std": 0.11004206538200378, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.13143374025821686, "rewards/total_composite/std": 0.11004206538200378, "reward": 0.13143374025821686, "reward_std": 0.11004206538200378, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03102065436542034, "sampling/sampling_logp_difference/max": 1.9875779151916504, "sampling/importance_sampling_ratio/min": 0.13702690601348877, "sampling/importance_sampling_ratio/mean": 0.9923664331436157, "sampling/importance_sampling_ratio/max": 1.671715497970581, "entropy": 0.1240494973026216, "clip_ratio/low_mean": 0.009703947464004159, "clip_ratio/low_min": 0.009703947464004159, "clip_ratio/high_mean": 0.00995732587762177, "clip_ratio/high_max": 0.00995732587762177, "clip_ratio/region_mean": 0.01966127334162593, "reward_total_mean": 0.13143374025821686, "reward_meter_mean": 0.13143374025821686, "reward_meter_std": 0.11004206538200378, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.13143374025821686, "reward_total_composite_std": 0.11004206538200378, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1635.0} {"timestamp_utc": "2026-04-11T22:58:32Z", "mode": "train", "global_step": 803, "epoch": 0.03225288187331807, "loss": 0.0049, "grad_norm": 4.46676778793335, "learning_rate": 7.56969696969697e-06, "num_tokens": 1775325.0, "completions/mean_length": 70.75, "completions/min_length": 70.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.75, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9594069123268127, "rewards/meter/std": 0.05366376414895058, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9594069123268127, "rewards/total_composite/std": 0.05366376414895058, "reward": 0.9594069123268127, "reward_std": 0.05366375669836998, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03665493428707123, "sampling/sampling_logp_difference/max": 1.9053394794464111, "sampling/importance_sampling_ratio/min": 0.14877213537693024, "sampling/importance_sampling_ratio/mean": 0.9985708594322205, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1510819010436535, "clip_ratio/low_mean": 0.005281690042465925, "clip_ratio/low_min": 0.005281690042465925, "clip_ratio/high_mean": 0.026509752846322954, "clip_ratio/high_max": 0.026509752846322954, "clip_ratio/region_mean": 0.03179144288878888, "reward_total_mean": 0.9594069123268127, "reward_meter_mean": 0.9594069123268127, "reward_meter_std": 0.05366376414895058, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9594069123268127, "reward_total_composite_std": 0.05366376414895058, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1636.0} {"timestamp_utc": "2026-04-11T22:58:44Z", "mode": "train", "global_step": 804, "epoch": 0.03229304735510302, "loss": 0.0506, "grad_norm": 0.5559709072113037, "learning_rate": 7.566666666666667e-06, "num_tokens": 1780007.0, "completions/mean_length": 505.25, "completions/min_length": 493.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 503.0, "completions/min_terminated_length": 493.0, "completions/max_terminated_length": 508.0, "rewards/meter/mean": 0.9569629430770874, "rewards/meter/std": 0.102637879550457, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.0235702246427536, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4376780688762665, "rewards/repeat_penalty/std": 0.22492921352386475, "rewards/total_composite/mean": 0.38479870557785034, "rewards/total_composite/std": 0.20677605271339417, "reward": 0.38479870557785034, "reward_std": 0.20677603781223297, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006570629775524139, "sampling/sampling_logp_difference/max": 2.7531256675720215, "sampling/importance_sampling_ratio/min": 0.06372835487127304, "sampling/importance_sampling_ratio/mean": 1.0004603862762451, "sampling/importance_sampling_ratio/max": 1.9659916162490845, "entropy": 0.014583299867808819, "clip_ratio/low_mean": 0.0004926113178953528, "clip_ratio/low_min": 0.0004926113178953528, "clip_ratio/high_mean": 0.003243225917685777, "clip_ratio/high_max": 0.003243225917685777, "clip_ratio/region_mean": 0.00373583723558113, "reward_total_mean": 0.38479870557785034, "reward_meter_mean": 0.9569629430770874, "reward_meter_std": 0.102637879550457, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.0235702246427536, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4376780688762665, "reward_repeat_penalty_std": 0.22492921352386475, "reward_total_composite_mean": 0.38479870557785034, "reward_total_composite_std": 0.20677605271339417, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1637.0} {"timestamp_utc": "2026-04-11T22:58:48Z", "mode": "train", "global_step": 805, "epoch": 0.032333212836887976, "loss": 0.0024, "grad_norm": 4.009762287139893, "learning_rate": 7.563636363636364e-06, "num_tokens": 1781817.0, "completions/mean_length": 67.25, "completions/min_length": 67.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.7574520111083984, "rewards/meter/std": 0.03009505569934845, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7574520111083984, "rewards/total_composite/std": 0.03009505569934845, "reward": 0.7574520111083984, "reward_std": 0.0300950538367033, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020206134766340256, "sampling/sampling_logp_difference/max": 1.227433443069458, "sampling/importance_sampling_ratio/min": 0.29304373264312744, "sampling/importance_sampling_ratio/mean": 0.9979568719863892, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05657489877194166, "clip_ratio/low_mean": 0.011085874866694212, "clip_ratio/low_min": 0.011085874866694212, "clip_ratio/high_mean": 0.00932835799176246, "clip_ratio/high_max": 0.00932835799176246, "clip_ratio/region_mean": 0.02041423285845667, "reward_total_mean": 0.7574520111083984, "reward_meter_mean": 0.7574520111083984, "reward_meter_std": 0.03009505569934845, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7574520111083984, "reward_total_composite_std": 0.03009505569934845, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1638.0} {"timestamp_utc": "2026-04-11T22:58:53Z", "mode": "train", "global_step": 806, "epoch": 0.03237337831867293, "loss": 0.0021, "grad_norm": 3.6810977458953857, "learning_rate": 7.560606060606062e-06, "num_tokens": 1783594.0, "completions/mean_length": 58.125, "completions/min_length": 57.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.990125298500061, "rewards/meter/std": 0.0014512698398903012, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.8252000212669373, "rewards/total_composite/std": 0.17693018913269043, "reward": 0.8252000212669373, "reward_std": 0.17693018913269043, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023440338671207428, "sampling/sampling_logp_difference/max": 1.1792564392089844, "sampling/importance_sampling_ratio/min": 0.30750730633735657, "sampling/importance_sampling_ratio/mean": 0.9979320168495178, "sampling/importance_sampling_ratio/max": 1.3977869749069214, "entropy": 0.11172830406576395, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.01062974869273603, "clip_ratio/high_max": 0.01062974869273603, "clip_ratio/region_mean": 0.01494009350426495, "reward_total_mean": 0.8252000212669373, "reward_meter_mean": 0.990125298500061, "reward_meter_std": 0.0014512698398903012, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.8252000212669373, "reward_total_composite_std": 0.17693018913269043, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1639.0} {"timestamp_utc": "2026-04-11T22:58:58Z", "mode": "train", "global_step": 807, "epoch": 0.032413543800457884, "loss": -0.0084, "grad_norm": 5.640214920043945, "learning_rate": 7.557575757575758e-06, "num_tokens": 1785577.0, "completions/mean_length": 75.875, "completions/min_length": 72.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7918260097503662, "rewards/meter/std": 0.3541868031024933, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7303093671798706, "rewards/total_composite/std": 0.35869720578193665, "reward": 0.7303093671798706, "reward_std": 0.35869717597961426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02571277692914009, "sampling/sampling_logp_difference/max": 1.1890771389007568, "sampling/importance_sampling_ratio/min": 0.4486599266529083, "sampling/importance_sampling_ratio/mean": 1.0031059980392456, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12689008563756943, "clip_ratio/low_mean": 0.00854225957300514, "clip_ratio/low_min": 0.00854225957300514, "clip_ratio/high_mean": 0.008098726975731552, "clip_ratio/high_max": 0.008098726975731552, "clip_ratio/region_mean": 0.01664098654873669, "reward_total_mean": 0.7303093671798706, "reward_meter_mean": 0.7918260097503662, "reward_meter_std": 0.3541868031024933, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7303093671798706, "reward_total_composite_std": 0.35869720578193665, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1640.0} {"timestamp_utc": "2026-04-11T22:59:02Z", "mode": "train", "global_step": 808, "epoch": 0.03245370928224284, "loss": -0.0018, "grad_norm": 14.179214477539062, "learning_rate": 7.5545454545454555e-06, "num_tokens": 1787137.0, "completions/mean_length": 38.0, "completions/min_length": 38.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8494601845741272, "rewards/meter/std": 0.3245563507080078, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8494601845741272, "rewards/total_composite/std": 0.3245563507080078, "reward": 0.8494601845741272, "reward_std": 0.3245563209056854, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026692412793636322, "sampling/sampling_logp_difference/max": 2.183236598968506, "sampling/importance_sampling_ratio/min": 0.11267625540494919, "sampling/importance_sampling_ratio/mean": 0.9977914690971375, "sampling/importance_sampling_ratio/max": 1.5366783142089844, "entropy": 0.09128655772656202, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.016447368543595076, "clip_ratio/high_max": 0.016447368543595076, "clip_ratio/region_mean": 0.016447368543595076, "reward_total_mean": 0.8494601845741272, "reward_meter_mean": 0.8494601845741272, "reward_meter_std": 0.3245563507080078, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8494601845741272, "reward_total_composite_std": 0.3245563507080078, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1641.0} {"timestamp_utc": "2026-04-11T22:59:09Z", "mode": "train", "global_step": 809, "epoch": 0.03249387476402779, "loss": -0.0031, "grad_norm": 1.1023591756820679, "learning_rate": 7.551515151515152e-06, "num_tokens": 1790539.0, "completions/mean_length": 218.25, "completions/min_length": 206.0, "completions/max_length": 230.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 218.25, "completions/min_terminated_length": 206.0, "completions/max_terminated_length": 230.0, "rewards/meter/mean": 0.9922986030578613, "rewards/meter/std": 0.004149852320551872, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5681818723678589, "rewards/repeat_penalty/std": 0.1735115498304367, "rewards/total_composite/mean": 0.5638895034790039, "rewards/total_composite/std": 0.17219732701778412, "reward": 0.5638895034790039, "reward_std": 0.17219732701778412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011151120997965336, "sampling/sampling_logp_difference/max": 1.5731306076049805, "sampling/importance_sampling_ratio/min": 0.20739488303661346, "sampling/importance_sampling_ratio/mean": 1.0026495456695557, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03606141824275255, "clip_ratio/low_mean": 0.005849295761436224, "clip_ratio/low_min": 0.005849295761436224, "clip_ratio/high_mean": 0.003968499368056655, "clip_ratio/high_max": 0.003968499368056655, "clip_ratio/region_mean": 0.009817795129492879, "reward_total_mean": 0.5638895034790039, "reward_meter_mean": 0.9922986030578613, "reward_meter_std": 0.004149852320551872, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5681818723678589, "reward_repeat_penalty_std": 0.1735115498304367, "reward_total_composite_mean": 0.5638895034790039, "reward_total_composite_std": 0.17219732701778412, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1642.0} {"timestamp_utc": "2026-04-11T22:59:14Z", "mode": "train", "global_step": 810, "epoch": 0.032534040245812745, "loss": 0.0017, "grad_norm": 5.80548095703125, "learning_rate": 7.548484848484849e-06, "num_tokens": 1792366.0, "completions/mean_length": 71.375, "completions/min_length": 70.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9742876887321472, "rewards/meter/std": 0.029914017766714096, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9332020878791809, "rewards/total_composite/std": 0.11529982835054398, "reward": 0.9332020878791809, "reward_std": 0.11529984325170517, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0373886339366436, "sampling/sampling_logp_difference/max": 1.9366556406021118, "sampling/importance_sampling_ratio/min": 0.14418534934520721, "sampling/importance_sampling_ratio/mean": 1.0060287714004517, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15044390503317118, "clip_ratio/low_mean": 0.005306840990670025, "clip_ratio/low_min": 0.005306840990670025, "clip_ratio/high_mean": 0.03143186215311289, "clip_ratio/high_max": 0.03143186215311289, "clip_ratio/region_mean": 0.036738703143782914, "reward_total_mean": 0.9332020878791809, "reward_meter_mean": 0.9742876887321472, "reward_meter_std": 0.029914017766714096, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9332020878791809, "reward_total_composite_std": 0.11529982835054398, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1643.0} {"timestamp_utc": "2026-04-11T22:59:20Z", "mode": "train", "global_step": 811, "epoch": 0.0325742057275977, "loss": 0.0192, "grad_norm": 2.172598123550415, "learning_rate": 7.545454545454546e-06, "num_tokens": 1795876.0, "completions/mean_length": 203.75, "completions/min_length": 196.0, "completions/max_length": 229.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 203.75, "completions/min_terminated_length": 196.0, "completions/max_terminated_length": 229.0, "rewards/meter/mean": 0.710443377494812, "rewards/meter/std": 0.2666179835796356, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4431818127632141, "rewards/repeat_penalty/std": 0.22498852014541626, "rewards/total_composite/mean": 0.3340369462966919, "rewards/total_composite/std": 0.2214316725730896, "reward": 0.3340369462966919, "reward_std": 0.2214316576719284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021875398233532906, "sampling/sampling_logp_difference/max": 4.384680271148682, "sampling/importance_sampling_ratio/min": 0.012466873973608017, "sampling/importance_sampling_ratio/mean": 1.0012760162353516, "sampling/importance_sampling_ratio/max": 1.9448559284210205, "entropy": 0.1404771413654089, "clip_ratio/low_mean": 0.006675369921140373, "clip_ratio/low_min": 0.006675369921140373, "clip_ratio/high_mean": 0.011296228156425059, "clip_ratio/high_max": 0.011296228156425059, "clip_ratio/region_mean": 0.01797159807756543, "reward_total_mean": 0.3340369462966919, "reward_meter_mean": 0.710443377494812, "reward_meter_std": 0.2666179835796356, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4431818127632141, "reward_repeat_penalty_std": 0.22498852014541626, "reward_total_composite_mean": 0.3340369462966919, "reward_total_composite_std": 0.2214316725730896, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1644.0} {"timestamp_utc": "2026-04-11T22:59:25Z", "mode": "train", "global_step": 812, "epoch": 0.03261437120938265, "loss": 0.0139, "grad_norm": 8.286386489868164, "learning_rate": 7.542424242424244e-06, "num_tokens": 1797566.0, "completions/mean_length": 65.25, "completions/min_length": 65.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.25, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9978810548782349, "rewards/meter/std": 0.0005103643052279949, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9563552141189575, "rewards/total_composite/std": 0.11796265840530396, "reward": 0.9563552141189575, "reward_std": 0.11796264350414276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007206229493021965, "sampling/sampling_logp_difference/max": 0.5880947113037109, "sampling/importance_sampling_ratio/min": 0.5553844571113586, "sampling/importance_sampling_ratio/mean": 1.002901315689087, "sampling/importance_sampling_ratio/max": 1.3499990701675415, "entropy": 0.036765412194654346, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005769230774603784, "clip_ratio/high_max": 0.005769230774603784, "clip_ratio/region_mean": 0.005769230774603784, "reward_total_mean": 0.9563552141189575, "reward_meter_mean": 0.9978810548782349, "reward_meter_std": 0.0005103643052279949, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9563552141189575, "reward_total_composite_std": 0.11796265840530396, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1645.0} {"timestamp_utc": "2026-04-11T22:59:34Z", "mode": "train", "global_step": 813, "epoch": 0.03265453669116761, "loss": 0.0079, "grad_norm": 1.0310794115066528, "learning_rate": 7.53939393939394e-06, "num_tokens": 1801761.0, "completions/mean_length": 332.375, "completions/min_length": 302.0, "completions/max_length": 353.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 332.375, "completions/min_terminated_length": 302.0, "completions/max_terminated_length": 353.0, "rewards/meter/mean": 0.9938352704048157, "rewards/meter/std": 0.008550022728741169, "rewards/count_adherence/mean": 0.890625, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5850183963775635, "rewards/repeat_penalty/std": 0.009098809212446213, "rewards/total_composite/mean": 0.517345666885376, "rewards/total_composite/std": 0.012124452739953995, "reward": 0.517345666885376, "reward_std": 0.012124458327889442, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0052232746966183186, "sampling/sampling_logp_difference/max": 0.7512289881706238, "sampling/importance_sampling_ratio/min": 0.47178637981414795, "sampling/importance_sampling_ratio/mean": 1.0009092092514038, "sampling/importance_sampling_ratio/max": 1.6740483045578003, "entropy": 0.02348946128040552, "clip_ratio/low_mean": 0.005218350415816531, "clip_ratio/low_min": 0.005218350415816531, "clip_ratio/high_mean": 0.0019113150192424655, "clip_ratio/high_max": 0.0019113150192424655, "clip_ratio/region_mean": 0.007129665435058996, "reward_total_mean": 0.517345666885376, "reward_meter_mean": 0.9938352704048157, "reward_meter_std": 0.008550022728741169, "reward_count_adherence_mean": 0.890625, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5850183963775635, "reward_repeat_penalty_std": 0.009098809212446213, "reward_total_composite_mean": 0.517345666885376, "reward_total_composite_std": 0.012124452739953995, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1646.0} {"timestamp_utc": "2026-04-11T22:59:40Z", "mode": "train", "global_step": 814, "epoch": 0.03269470217295257, "loss": -0.001, "grad_norm": 13.524036407470703, "learning_rate": 7.536363636363637e-06, "num_tokens": 1804516.0, "completions/mean_length": 159.375, "completions/min_length": 158.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.375, "completions/min_terminated_length": 158.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.9979825019836426, "rewards/meter/std": 0.0006693408940918744, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6653216481208801, "rewards/total_composite/std": 0.00044622019049711525, "reward": 0.6653216481208801, "reward_std": 0.00044622053974308074, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005735776387155056, "sampling/sampling_logp_difference/max": 2.232571601867676, "sampling/importance_sampling_ratio/min": 0.10725226998329163, "sampling/importance_sampling_ratio/mean": 0.9996488094329834, "sampling/importance_sampling_ratio/max": 1.493585467338562, "entropy": 0.03128327080048621, "clip_ratio/low_mean": 0.00237341778120026, "clip_ratio/low_min": 0.00237341778120026, "clip_ratio/high_mean": 0.00237341778120026, "clip_ratio/high_max": 0.00237341778120026, "clip_ratio/region_mean": 0.00474683556240052, "reward_total_mean": 0.6653216481208801, "reward_meter_mean": 0.9979825019836426, "reward_meter_std": 0.0006693408940918744, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6653216481208801, "reward_total_composite_std": 0.00044622019049711525, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1647.0} {"timestamp_utc": "2026-04-11T22:59:44Z", "mode": "train", "global_step": 815, "epoch": 0.03273486765473752, "loss": 0.0458, "grad_norm": 8.952130317687988, "learning_rate": 7.533333333333334e-06, "num_tokens": 1806317.0, "completions/mean_length": 62.125, "completions/min_length": 54.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9308251142501831, "rewards/meter/std": 0.15261007845401764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.8894620537757874, "rewards/total_composite/std": 0.17642506957054138, "reward": 0.8894620537757874, "reward_std": 0.1764250546693802, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024382825940847397, "sampling/sampling_logp_difference/max": 1.424285888671875, "sampling/importance_sampling_ratio/min": 0.2406802922487259, "sampling/importance_sampling_ratio/mean": 0.9998781085014343, "sampling/importance_sampling_ratio/max": 1.5671416521072388, "entropy": 0.10703426506370306, "clip_ratio/low_mean": 0.011092530796304345, "clip_ratio/low_min": 0.011092530796304345, "clip_ratio/high_mean": 0.016365812392905354, "clip_ratio/high_max": 0.016365812392905354, "clip_ratio/region_mean": 0.0274583431892097, "reward_total_mean": 0.8894620537757874, "reward_meter_mean": 0.9308251142501831, "reward_meter_std": 0.15261007845401764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.8894620537757874, "reward_total_composite_std": 0.17642506957054138, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1648.0} {"timestamp_utc": "2026-04-11T22:59:49Z", "mode": "train", "global_step": 816, "epoch": 0.032775033136522476, "loss": -0.0066, "grad_norm": 8.190361022949219, "learning_rate": 7.530303030303031e-06, "num_tokens": 1807821.0, "completions/mean_length": 39.0, "completions/min_length": 38.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.7446458339691162, "rewards/meter/std": 0.30041757225990295, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7446458339691162, "rewards/total_composite/std": 0.30041757225990295, "reward": 0.7446458339691162, "reward_std": 0.30041757225990295, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.059971921145915985, "sampling/sampling_logp_difference/max": 2.114863395690918, "sampling/importance_sampling_ratio/min": 0.12064976990222931, "sampling/importance_sampling_ratio/mean": 0.9998385310173035, "sampling/importance_sampling_ratio/max": 1.8297884464263916, "entropy": 0.16819044947624207, "clip_ratio/low_mean": 0.023026316426694393, "clip_ratio/low_min": 0.023026316426694393, "clip_ratio/high_mean": 0.028449730249121785, "clip_ratio/high_max": 0.028449730249121785, "clip_ratio/region_mean": 0.05147604667581618, "reward_total_mean": 0.7446458339691162, "reward_meter_mean": 0.7446458339691162, "reward_meter_std": 0.30041757225990295, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7446458339691162, "reward_total_composite_std": 0.30041757225990295, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1649.0} {"timestamp_utc": "2026-04-11T22:59:54Z", "mode": "train", "global_step": 817, "epoch": 0.03281519861830743, "loss": 0.0413, "grad_norm": 3.7302210330963135, "learning_rate": 7.5272727272727274e-06, "num_tokens": 1810070.0, "completions/mean_length": 107.125, "completions/min_length": 102.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.125, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.3294169008731842, "rewards/meter/std": 0.28731057047843933, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7750000357627869, "rewards/repeat_penalty/std": 0.12817399203777313, "rewards/total_composite/mean": 0.2755756676197052, "rewards/total_composite/std": 0.2410556823015213, "reward": 0.2755756676197052, "reward_std": 0.2410556823015213, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01802036352455616, "sampling/sampling_logp_difference/max": 1.4399476051330566, "sampling/importance_sampling_ratio/min": 0.2369401901960373, "sampling/importance_sampling_ratio/mean": 1.0044699907302856, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05588802928104997, "clip_ratio/low_mean": 0.01120058260858059, "clip_ratio/low_min": 0.01120058260858059, "clip_ratio/high_mean": 0.013432800536975265, "clip_ratio/high_max": 0.013432800536975265, "clip_ratio/region_mean": 0.024633383145555854, "reward_total_mean": 0.2755756676197052, "reward_meter_mean": 0.3294169008731842, "reward_meter_std": 0.28731057047843933, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7750000357627869, "reward_repeat_penalty_std": 0.12817399203777313, "reward_total_composite_mean": 0.2755756676197052, "reward_total_composite_std": 0.2410556823015213, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1650.0} {"timestamp_utc": "2026-04-11T23:00:03Z", "mode": "train", "global_step": 818, "epoch": 0.032855364100092384, "loss": 0.0108, "grad_norm": 1.0928229093551636, "learning_rate": 7.524242424242425e-06, "num_tokens": 1814452.0, "completions/mean_length": 357.75, "completions/min_length": 346.0, "completions/max_length": 384.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 357.75, "completions/min_terminated_length": 346.0, "completions/max_terminated_length": 384.0, "rewards/meter/mean": 0.9965507984161377, "rewards/meter/std": 0.0037117046304047108, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.05750546231865883, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5523655414581299, "rewards/repeat_penalty/std": 0.10810358822345734, "rewards/total_composite/mean": 0.5283541083335876, "rewards/total_composite/std": 0.11482103914022446, "reward": 0.5283541083335876, "reward_std": 0.11482104659080505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00884958729147911, "sampling/sampling_logp_difference/max": 1.062759518623352, "sampling/importance_sampling_ratio/min": 0.3455010652542114, "sampling/importance_sampling_ratio/mean": 1.0001471042633057, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03675817488692701, "clip_ratio/low_mean": 0.003796361561398953, "clip_ratio/low_min": 0.003796361561398953, "clip_ratio/high_mean": 0.006020042230375111, "clip_ratio/high_max": 0.006020042230375111, "clip_ratio/region_mean": 0.009816403791774064, "reward_total_mean": 0.5283541083335876, "reward_meter_mean": 0.9965507984161377, "reward_meter_std": 0.0037117046304047108, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.05750546231865883, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5523655414581299, "reward_repeat_penalty_std": 0.10810358822345734, "reward_total_composite_mean": 0.5283541083335876, "reward_total_composite_std": 0.11482103914022446, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1651.0} {"timestamp_utc": "2026-04-11T23:00:08Z", "mode": "train", "global_step": 819, "epoch": 0.03289552958187734, "loss": 0.0279, "grad_norm": 6.570438861846924, "learning_rate": 7.521212121212121e-06, "num_tokens": 1816193.0, "completions/mean_length": 61.625, "completions/min_length": 59.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9746619462966919, "rewards/meter/std": 0.020627282559871674, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9746619462966919, "rewards/total_composite/std": 0.020627282559871674, "reward": 0.9746619462966919, "reward_std": 0.020627308636903763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02879003807902336, "sampling/sampling_logp_difference/max": 1.3429758548736572, "sampling/importance_sampling_ratio/min": 0.26106762886047363, "sampling/importance_sampling_ratio/mean": 0.9997745156288147, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10824673250317574, "clip_ratio/low_mean": 0.013770792167633772, "clip_ratio/low_min": 0.013770792167633772, "clip_ratio/high_mean": 0.022682114504277706, "clip_ratio/high_max": 0.022682114504277706, "clip_ratio/region_mean": 0.03645290667191148, "reward_total_mean": 0.9746619462966919, "reward_meter_mean": 0.9746619462966919, "reward_meter_std": 0.020627282559871674, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9746619462966919, "reward_total_composite_std": 0.020627282559871674, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1652.0} {"timestamp_utc": "2026-04-11T23:00:14Z", "mode": "train", "global_step": 820, "epoch": 0.03293569506366229, "loss": 0.0185, "grad_norm": 2.3457443714141846, "learning_rate": 7.518181818181819e-06, "num_tokens": 1818812.0, "completions/mean_length": 149.375, "completions/min_length": 143.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.375, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.7805307507514954, "rewards/meter/std": 0.1253480166196823, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5714285373687744, "rewards/repeat_penalty/std": 0.17074695229530334, "rewards/total_composite/mean": 0.4300089478492737, "rewards/total_composite/std": 0.0887664407491684, "reward": 0.4300089478492737, "reward_std": 0.0887664407491684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015637392178177834, "sampling/sampling_logp_difference/max": 1.2027087211608887, "sampling/importance_sampling_ratio/min": 0.3003794550895691, "sampling/importance_sampling_ratio/mean": 1.0010299682617188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06724543264135718, "clip_ratio/low_mean": 0.007311595429200679, "clip_ratio/low_min": 0.007311595429200679, "clip_ratio/high_mean": 0.008506174897775054, "clip_ratio/high_max": 0.008506174897775054, "clip_ratio/region_mean": 0.015817770326975733, "reward_total_mean": 0.4300089478492737, "reward_meter_mean": 0.7805307507514954, "reward_meter_std": 0.1253480166196823, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5714285373687744, "reward_repeat_penalty_std": 0.17074695229530334, "reward_total_composite_mean": 0.4300089478492737, "reward_total_composite_std": 0.0887664407491684, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1653.0} {"timestamp_utc": "2026-04-11T23:00:23Z", "mode": "train", "global_step": 821, "epoch": 0.032975860545447246, "loss": -0.1471, "grad_norm": 0.9305291175842285, "learning_rate": 7.515151515151516e-06, "num_tokens": 1820662.0, "completions/mean_length": 190.25, "completions/min_length": 77.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 83.0, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.5668963193893433, "rewards/meter/std": 0.39650967717170715, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5668963193893433, "rewards/total_composite/std": 0.39650967717170715, "reward": 0.5668963193893433, "reward_std": 0.39650964736938477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019788384437561035, "sampling/sampling_logp_difference/max": 0.7466448545455933, "sampling/importance_sampling_ratio/min": 0.6215139627456665, "sampling/importance_sampling_ratio/mean": 1.0027402639389038, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07134009897708893, "clip_ratio/low_mean": 0.007159017724916339, "clip_ratio/low_min": 0.007159017724916339, "clip_ratio/high_mean": 0.01558885129634291, "clip_ratio/high_max": 0.01558885129634291, "clip_ratio/region_mean": 0.022747869021259248, "reward_total_mean": 0.5668963193893433, "reward_meter_mean": 0.5668963193893433, "reward_meter_std": 0.39650967717170715, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5668963193893433, "reward_total_composite_std": 0.39650967717170715, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1654.0} {"timestamp_utc": "2026-04-11T23:00:28Z", "mode": "train", "global_step": 822, "epoch": 0.0330160260272322, "loss": 0.0105, "grad_norm": 3.704124689102173, "learning_rate": 7.512121212121213e-06, "num_tokens": 1822669.0, "completions/mean_length": 77.875, "completions/min_length": 75.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.875, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.8737984299659729, "rewards/meter/std": 0.14763900637626648, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8737984299659729, "rewards/total_composite/std": 0.14763900637626648, "reward": 0.8737984299659729, "reward_std": 0.14763899147510529, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018405567854642868, "sampling/sampling_logp_difference/max": 1.2704942226409912, "sampling/importance_sampling_ratio/min": 0.28069284558296204, "sampling/importance_sampling_ratio/mean": 1.0058021545410156, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09962678607553244, "clip_ratio/low_mean": 0.006382113788276911, "clip_ratio/low_min": 0.006382113788276911, "clip_ratio/high_mean": 0.009740259731188416, "clip_ratio/high_max": 0.009740259731188416, "clip_ratio/region_mean": 0.016122373519465327, "reward_total_mean": 0.8737984299659729, "reward_meter_mean": 0.8737984299659729, "reward_meter_std": 0.14763900637626648, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8737984299659729, "reward_total_composite_std": 0.14763900637626648, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1655.0} {"timestamp_utc": "2026-04-11T23:00:35Z", "mode": "train", "global_step": 823, "epoch": 0.033056191509017153, "loss": 0.0055, "grad_norm": 2.836785078048706, "learning_rate": 7.509090909090909e-06, "num_tokens": 1826360.0, "completions/mean_length": 284.375, "completions/min_length": 272.0, "completions/max_length": 297.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 284.375, "completions/min_terminated_length": 272.0, "completions/max_terminated_length": 297.0, "rewards/meter/mean": 0.9983997344970703, "rewards/meter/std": 0.0007759786094538867, "rewards/count_adherence/mean": 0.9821428656578064, "rewards/count_adherence/std": 0.05050762742757797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5952796936035156, "rewards/repeat_penalty/std": 0.08133196830749512, "rewards/total_composite/mean": 0.5862284898757935, "rewards/total_composite/std": 0.09859947860240936, "reward": 0.5862284898757935, "reward_std": 0.09859946370124817, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015353661961853504, "sampling/sampling_logp_difference/max": 2.6265687942504883, "sampling/importance_sampling_ratio/min": 0.07232620567083359, "sampling/importance_sampling_ratio/mean": 1.0005059242248535, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03474967950023711, "clip_ratio/low_mean": 0.002257478976389393, "clip_ratio/low_min": 0.002257478976389393, "clip_ratio/high_mean": 0.0065078792395070195, "clip_ratio/high_max": 0.0065078792395070195, "clip_ratio/region_mean": 0.008765358215896413, "reward_total_mean": 0.5862284898757935, "reward_meter_mean": 0.9983997344970703, "reward_meter_std": 0.0007759786094538867, "reward_count_adherence_mean": 0.9821428656578064, "reward_count_adherence_std": 0.05050762742757797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5952796936035156, "reward_repeat_penalty_std": 0.08133196830749512, "reward_total_composite_mean": 0.5862284898757935, "reward_total_composite_std": 0.09859947860240936, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1656.0} {"timestamp_utc": "2026-04-11T23:00:40Z", "mode": "train", "global_step": 824, "epoch": 0.03309635699080211, "loss": 0.0168, "grad_norm": 4.631432056427002, "learning_rate": 7.5060606060606065e-06, "num_tokens": 1828232.0, "completions/mean_length": 71.0, "completions/min_length": 70.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7449524998664856, "rewards/meter/std": 0.04608132317662239, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7449524998664856, "rewards/total_composite/std": 0.04608132317662239, "reward": 0.7449524998664856, "reward_std": 0.04608132690191269, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01667422614991665, "sampling/sampling_logp_difference/max": 2.1282317638397217, "sampling/importance_sampling_ratio/min": 0.1190476045012474, "sampling/importance_sampling_ratio/mean": 0.9986546039581299, "sampling/importance_sampling_ratio/max": 1.426641583442688, "entropy": 0.04505776287987828, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/high_mean": 0.0070436508394777775, "clip_ratio/high_max": 0.0070436508394777775, "clip_ratio/region_mean": 0.008779761963523924, "reward_total_mean": 0.7449524998664856, "reward_meter_mean": 0.7449524998664856, "reward_meter_std": 0.04608132317662239, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7449524998664856, "reward_total_composite_std": 0.04608132317662239, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1657.0} {"timestamp_utc": "2026-04-11T23:00:45Z", "mode": "train", "global_step": 825, "epoch": 0.03313652247258706, "loss": 0.0481, "grad_norm": 4.750690937042236, "learning_rate": 7.503030303030303e-06, "num_tokens": 1830123.0, "completions/mean_length": 68.375, "completions/min_length": 64.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.4837471544742584, "rewards/meter/std": 0.22915734350681305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4837471544742584, "rewards/total_composite/std": 0.22915734350681305, "reward": 0.4837471544742584, "reward_std": 0.22915734350681305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031915146857500076, "sampling/sampling_logp_difference/max": 1.8651485443115234, "sampling/importance_sampling_ratio/min": 0.15487320721149445, "sampling/importance_sampling_ratio/mean": 0.9999799728393555, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10288840066641569, "clip_ratio/low_mean": 0.024154700804501772, "clip_ratio/low_min": 0.024154700804501772, "clip_ratio/high_mean": 0.028171515092253685, "clip_ratio/high_max": 0.028171515092253685, "clip_ratio/region_mean": 0.05232621589675546, "reward_total_mean": 0.4837471544742584, "reward_meter_mean": 0.4837471544742584, "reward_meter_std": 0.22915734350681305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4837471544742584, "reward_total_composite_std": 0.22915734350681305, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1658.0} {"timestamp_utc": "2026-04-11T23:00:54Z", "mode": "train", "global_step": 826, "epoch": 0.033176687954372015, "loss": 0.0159, "grad_norm": 1.1773563623428345, "learning_rate": 7.500000000000001e-06, "num_tokens": 1835244.0, "completions/mean_length": 400.125, "completions/min_length": 375.0, "completions/max_length": 421.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 400.125, "completions/min_terminated_length": 375.0, "completions/max_terminated_length": 421.0, "rewards/meter/mean": 0.7846977710723877, "rewards/meter/std": 0.39648592472076416, "rewards/count_adherence/mean": 0.9545454978942871, "rewards/count_adherence/std": 0.0485929399728775, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5507364273071289, "rewards/repeat_penalty/std": 0.050370436161756516, "rewards/total_composite/mean": 0.3971686363220215, "rewards/total_composite/std": 0.1935662180185318, "reward": 0.3971686363220215, "reward_std": 0.1935662031173706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007981406524777412, "sampling/sampling_logp_difference/max": 1.5345345735549927, "sampling/importance_sampling_ratio/min": 0.25283992290496826, "sampling/importance_sampling_ratio/mean": 1.0009510517120361, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0334128841641359, "clip_ratio/low_mean": 0.0033315176842734218, "clip_ratio/low_min": 0.0033315176842734218, "clip_ratio/high_mean": 0.005249909590929747, "clip_ratio/high_max": 0.005249909590929747, "clip_ratio/region_mean": 0.008581427275203168, "reward_total_mean": 0.3971686363220215, "reward_meter_mean": 0.7846977710723877, "reward_meter_std": 0.39648592472076416, "reward_count_adherence_mean": 0.9545454978942871, "reward_count_adherence_std": 0.0485929399728775, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5507364273071289, "reward_repeat_penalty_std": 0.050370436161756516, "reward_total_composite_mean": 0.3971686363220215, "reward_total_composite_std": 0.1935662180185318, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1659.0} {"timestamp_utc": "2026-04-11T23:00:59Z", "mode": "train", "global_step": 827, "epoch": 0.03321685343615697, "loss": -0.0012, "grad_norm": 3.7937092781066895, "learning_rate": 7.496969696969698e-06, "num_tokens": 1837278.0, "completions/mean_length": 68.25, "completions/min_length": 68.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.15423895418643951, "rewards/meter/std": 0.07883358746767044, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.15423895418643951, "rewards/total_composite/std": 0.07883358746767044, "reward": 0.15423895418643951, "reward_std": 0.07883358746767044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016422174870967865, "sampling/sampling_logp_difference/max": 0.6457552909851074, "sampling/importance_sampling_ratio/min": 0.5242664217948914, "sampling/importance_sampling_ratio/mean": 0.9993041157722473, "sampling/importance_sampling_ratio/max": 1.73274564743042, "entropy": 0.05680654477328062, "clip_ratio/low_mean": 0.009191176504828036, "clip_ratio/low_min": 0.009191176504828036, "clip_ratio/high_mean": 0.005434782709926367, "clip_ratio/high_max": 0.005434782709926367, "clip_ratio/region_mean": 0.014625959214754403, "reward_total_mean": 0.15423895418643951, "reward_meter_mean": 0.15423895418643951, "reward_meter_std": 0.07883358746767044, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.15423895418643951, "reward_total_composite_std": 0.07883358746767044, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1660.0} {"timestamp_utc": "2026-04-11T23:01:04Z", "mode": "train", "global_step": 828, "epoch": 0.03325701891794192, "loss": 0.0145, "grad_norm": 6.0185723304748535, "learning_rate": 7.493939393939395e-06, "num_tokens": 1839058.0, "completions/mean_length": 63.5, "completions/min_length": 61.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9804123640060425, "rewards/meter/std": 0.014910156838595867, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9804123640060425, "rewards/total_composite/std": 0.014910156838595867, "reward": 0.9804123640060425, "reward_std": 0.014910157769918442, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028718305751681328, "sampling/sampling_logp_difference/max": 1.3754212856292725, "sampling/importance_sampling_ratio/min": 0.25273311138153076, "sampling/importance_sampling_ratio/mean": 1.0006060600280762, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11623660661280155, "clip_ratio/low_mean": 0.0038470644503831863, "clip_ratio/low_min": 0.0038470644503831863, "clip_ratio/high_mean": 0.01206992007791996, "clip_ratio/high_max": 0.01206992007791996, "clip_ratio/region_mean": 0.015916984528303146, "reward_total_mean": 0.9804123640060425, "reward_meter_mean": 0.9804123640060425, "reward_meter_std": 0.014910156838595867, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9804123640060425, "reward_total_composite_std": 0.014910156838595867, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1661.0} {"timestamp_utc": "2026-04-11T23:01:10Z", "mode": "train", "global_step": 829, "epoch": 0.03329718439972688, "loss": 0.0432, "grad_norm": 1.7107356786727905, "learning_rate": 7.490909090909092e-06, "num_tokens": 1841757.0, "completions/mean_length": 173.375, "completions/min_length": 151.0, "completions/max_length": 186.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 173.375, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.9961064457893372, "rewards/meter/std": 0.0029631657525897026, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.625, "rewards/repeat_penalty/std": 0.16967642307281494, "rewards/total_composite/mean": 0.6229597926139832, "rewards/total_composite/std": 0.17044374346733093, "reward": 0.6229597926139832, "reward_std": 0.17044374346733093, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017439143732190132, "sampling/sampling_logp_difference/max": 1.495316505432129, "sampling/importance_sampling_ratio/min": 0.22417764365673065, "sampling/importance_sampling_ratio/mean": 0.9979692101478577, "sampling/importance_sampling_ratio/max": 1.7753796577453613, "entropy": 0.05448292032815516, "clip_ratio/low_mean": 0.002016129030380398, "clip_ratio/low_min": 0.002016129030380398, "clip_ratio/high_mean": 0.016328177880495787, "clip_ratio/high_max": 0.016328177880495787, "clip_ratio/region_mean": 0.018344306910876185, "reward_total_mean": 0.6229597926139832, "reward_meter_mean": 0.9961064457893372, "reward_meter_std": 0.0029631657525897026, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.625, "reward_repeat_penalty_std": 0.16967642307281494, "reward_total_composite_mean": 0.6229597926139832, "reward_total_composite_std": 0.17044374346733093, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1662.0} {"timestamp_utc": "2026-04-11T23:01:15Z", "mode": "train", "global_step": 830, "epoch": 0.03333734988151183, "loss": 0.034, "grad_norm": 12.176047325134277, "learning_rate": 7.487878787878788e-06, "num_tokens": 1843710.0, "completions/mean_length": 74.125, "completions/min_length": 68.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9026094079017639, "rewards/meter/std": 0.14784321188926697, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9026094079017639, "rewards/total_composite/std": 0.14784321188926697, "reward": 0.9026094079017639, "reward_std": 0.14784321188926697, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.060764994472265244, "sampling/sampling_logp_difference/max": 5.8967156410217285, "sampling/importance_sampling_ratio/min": 0.0027484570164233446, "sampling/importance_sampling_ratio/mean": 0.9942511916160583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12799056991934776, "clip_ratio/low_mean": 0.007117270142771304, "clip_ratio/low_min": 0.007117270142771304, "clip_ratio/high_mean": 0.019965628627687693, "clip_ratio/high_max": 0.019965628627687693, "clip_ratio/region_mean": 0.027082898770458996, "reward_total_mean": 0.9026094079017639, "reward_meter_mean": 0.9026094079017639, "reward_meter_std": 0.14784321188926697, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9026094079017639, "reward_total_composite_std": 0.14784321188926697, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1663.0} {"timestamp_utc": "2026-04-11T23:01:21Z", "mode": "train", "global_step": 831, "epoch": 0.033377515363296785, "loss": -0.003, "grad_norm": 4.106361389160156, "learning_rate": 7.484848484848486e-06, "num_tokens": 1845975.0, "completions/mean_length": 108.125, "completions/min_length": 105.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.125, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.3786888122558594, "rewards/meter/std": 0.35290807485580444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.3143823742866516, "rewards/total_composite/std": 0.282669335603714, "reward": 0.3143823742866516, "reward_std": 0.282669335603714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034197013825178146, "sampling/sampling_logp_difference/max": 2.057035207748413, "sampling/importance_sampling_ratio/min": 0.12783241271972656, "sampling/importance_sampling_ratio/mean": 0.9974236488342285, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12897953018546104, "clip_ratio/low_mean": 0.016522707068361342, "clip_ratio/low_min": 0.016522707068361342, "clip_ratio/high_mean": 0.01371849060524255, "clip_ratio/high_max": 0.01371849060524255, "clip_ratio/region_mean": 0.030241197673603892, "reward_total_mean": 0.3143823742866516, "reward_meter_mean": 0.3786888122558594, "reward_meter_std": 0.35290807485580444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.3143823742866516, "reward_total_composite_std": 0.282669335603714, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1664.0} {"timestamp_utc": "2026-04-11T23:01:32Z", "mode": "train", "global_step": 832, "epoch": 0.03341768084508174, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.481818181818182e-06, "num_tokens": 1847831.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.6360248923301697, "rewards/meter/std": 0.25572946667671204, "rewards/count_adherence/mean": 0.8602941036224365, "rewards/count_adherence/std": 0.0304440688341856, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5477695465087891, "rewards/repeat_penalty/std": 0.011733362451195717, "rewards/total_composite/mean": 0.3008078336715698, "rewards/total_composite/std": 0.12469936162233353, "reward": 0.3008078336715698, "reward_std": 0.12469936162233353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.3008078336715698, "reward_meter_mean": 0.6360248923301697, "reward_meter_std": 0.25572946667671204, "reward_count_adherence_mean": 0.8602941036224365, "reward_count_adherence_std": 0.0304440688341856, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5477695465087891, "reward_repeat_penalty_std": 0.011733362451195717, "reward_total_composite_mean": 0.3008078336715698, "reward_total_composite_std": 0.12469936162233353, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1665.0} {"timestamp_utc": "2026-04-11T23:01:36Z", "mode": "train", "global_step": 833, "epoch": 0.03345784632686669, "loss": 0.0048, "grad_norm": 8.433792114257812, "learning_rate": 7.47878787878788e-06, "num_tokens": 1849335.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9924166798591614, "rewards/meter/std": 0.00017309709801338613, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924166798591614, "rewards/total_composite/std": 0.00017309709801338613, "reward": 0.9924166798591614, "reward_std": 0.00017310312250629067, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008245354518294334, "sampling/sampling_logp_difference/max": 0.6460732221603394, "sampling/importance_sampling_ratio/min": 0.5240997672080994, "sampling/importance_sampling_ratio/mean": 1.0005563497543335, "sampling/importance_sampling_ratio/max": 1.1176583766937256, "entropy": 0.04187649488449097, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9924166798591614, "reward_meter_mean": 0.9924166798591614, "reward_meter_std": 0.00017309709801338613, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9924166798591614, "reward_total_composite_std": 0.00017309709801338613, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1666.0} {"timestamp_utc": "2026-04-11T23:01:41Z", "mode": "train", "global_step": 834, "epoch": 0.03349801180865165, "loss": 0.0342, "grad_norm": 7.6343560218811035, "learning_rate": 7.4757575757575765e-06, "num_tokens": 1851111.0, "completions/mean_length": 68.0, "completions/min_length": 62.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.40049153566360474, "rewards/meter/std": 0.3537753224372864, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.40049153566360474, "rewards/total_composite/std": 0.3537753224372864, "reward": 0.40049153566360474, "reward_std": 0.3537753224372864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0617922842502594, "sampling/sampling_logp_difference/max": 3.229440689086914, "sampling/importance_sampling_ratio/min": 0.03957962989807129, "sampling/importance_sampling_ratio/mean": 1.0015736818313599, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33059297781437635, "clip_ratio/low_mean": 0.035882155993022025, "clip_ratio/low_min": 0.035882155993022025, "clip_ratio/high_mean": 0.009836265817284584, "clip_ratio/high_max": 0.009836265817284584, "clip_ratio/region_mean": 0.04571842181030661, "reward_total_mean": 0.40049153566360474, "reward_meter_mean": 0.40049153566360474, "reward_meter_std": 0.3537753224372864, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.40049153566360474, "reward_total_composite_std": 0.3537753224372864, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1667.0} {"timestamp_utc": "2026-04-11T23:01:46Z", "mode": "train", "global_step": 835, "epoch": 0.0335381772904366, "loss": 0.0177, "grad_norm": 1.6731441020965576, "learning_rate": 7.472727272727274e-06, "num_tokens": 1853656.0, "completions/mean_length": 150.125, "completions/min_length": 149.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.125, "completions/min_terminated_length": 149.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.9238545894622803, "rewards/meter/std": 0.2039637267589569, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6159030199050903, "rewards/total_composite/std": 0.13597580790519714, "reward": 0.6159030199050903, "reward_std": 0.13597580790519714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00628118310123682, "sampling/sampling_logp_difference/max": 1.1030478477478027, "sampling/importance_sampling_ratio/min": 0.3318580687046051, "sampling/importance_sampling_ratio/mean": 0.9989848136901855, "sampling/importance_sampling_ratio/max": 1.6532901525497437, "entropy": 0.03071731375530362, "clip_ratio/low_mean": 0.0007961783558130264, "clip_ratio/low_min": 0.0007961783558130264, "clip_ratio/high_mean": 0.004172259592451155, "clip_ratio/high_max": 0.004172259592451155, "clip_ratio/region_mean": 0.004968437948264182, "reward_total_mean": 0.6159030199050903, "reward_meter_mean": 0.9238545894622803, "reward_meter_std": 0.2039637267589569, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6159030199050903, "reward_total_composite_std": 0.13597580790519714, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1668.0} {"timestamp_utc": "2026-04-11T23:01:52Z", "mode": "train", "global_step": 836, "epoch": 0.033578342772221555, "loss": -0.0007, "grad_norm": 2.8023297786712646, "learning_rate": 7.46969696969697e-06, "num_tokens": 1856194.0, "completions/mean_length": 142.25, "completions/min_length": 139.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.25, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.8496776819229126, "rewards/meter/std": 0.3326415419578552, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8214285373687744, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.7131145596504211, "rewards/total_composite/std": 0.2920505106449127, "reward": 0.7131145596504211, "reward_std": 0.29205048084259033, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012105772271752357, "sampling/sampling_logp_difference/max": 1.3084237575531006, "sampling/importance_sampling_ratio/min": 0.2702457010746002, "sampling/importance_sampling_ratio/mean": 0.9984580278396606, "sampling/importance_sampling_ratio/max": 1.5496021509170532, "entropy": 0.044805840123444796, "clip_ratio/low_mean": 0.001773406460415572, "clip_ratio/low_min": 0.001773406460415572, "clip_ratio/high_mean": 0.007069471699651331, "clip_ratio/high_max": 0.007069471699651331, "clip_ratio/region_mean": 0.008842878160066903, "reward_total_mean": 0.7131145596504211, "reward_meter_mean": 0.8496776819229126, "reward_meter_std": 0.3326415419578552, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8214285373687744, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.7131145596504211, "reward_total_composite_std": 0.2920505106449127, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1669.0} {"timestamp_utc": "2026-04-11T23:01:56Z", "mode": "train", "global_step": 837, "epoch": 0.03361850825400651, "loss": 0.0053, "grad_norm": 11.809463500976562, "learning_rate": 7.4666666666666675e-06, "num_tokens": 1857618.0, "completions/mean_length": 33.0, "completions/min_length": 33.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9952504634857178, "rewards/meter/std": 0.0004852505517192185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952504634857178, "rewards/total_composite/std": 0.0004852505517192185, "reward": 0.9952504634857178, "reward_std": 0.0004852571291849017, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017783869057893753, "sampling/sampling_logp_difference/max": 0.7500922679901123, "sampling/importance_sampling_ratio/min": 0.4723230004310608, "sampling/importance_sampling_ratio/mean": 1.0051133632659912, "sampling/importance_sampling_ratio/max": 1.5762113332748413, "entropy": 0.0944258663803339, "clip_ratio/low_mean": 0.01515151560306549, "clip_ratio/low_min": 0.01515151560306549, "clip_ratio/high_mean": 0.011363636702299118, "clip_ratio/high_max": 0.011363636702299118, "clip_ratio/region_mean": 0.02651515230536461, "reward_total_mean": 0.9952504634857178, "reward_meter_mean": 0.9952504634857178, "reward_meter_std": 0.0004852505517192185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9952504634857178, "reward_total_composite_std": 0.0004852505517192185, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1670.0} {"timestamp_utc": "2026-04-11T23:02:04Z", "mode": "train", "global_step": 838, "epoch": 0.03365867373579146, "loss": 0.0061, "grad_norm": 2.168991804122925, "learning_rate": 7.463636363636364e-06, "num_tokens": 1861717.0, "completions/mean_length": 295.375, "completions/min_length": 264.0, "completions/max_length": 331.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 295.375, "completions/min_terminated_length": 264.0, "completions/max_terminated_length": 331.0, "rewards/meter/mean": 0.869155764579773, "rewards/meter/std": 0.35062167048454285, "rewards/count_adherence/mean": 0.828125, "rewards/count_adherence/std": 0.06469365209341049, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5626935362815857, "rewards/repeat_penalty/std": 0.04896574467420578, "rewards/total_composite/mean": 0.4079222083091736, "rewards/total_composite/std": 0.1708296537399292, "reward": 0.4079222083091736, "reward_std": 0.170829638838768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009027719497680664, "sampling/sampling_logp_difference/max": 1.8500938415527344, "sampling/importance_sampling_ratio/min": 0.15722240507602692, "sampling/importance_sampling_ratio/mean": 0.9985241889953613, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02827131818048656, "clip_ratio/low_mean": 0.0017137863032985479, "clip_ratio/low_min": 0.0017137863032985479, "clip_ratio/high_mean": 0.004315092111937702, "clip_ratio/high_max": 0.004315092111937702, "clip_ratio/region_mean": 0.0060288784152362496, "reward_total_mean": 0.4079222083091736, "reward_meter_mean": 0.869155764579773, "reward_meter_std": 0.35062167048454285, "reward_count_adherence_mean": 0.828125, "reward_count_adherence_std": 0.06469365209341049, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5626935362815857, "reward_repeat_penalty_std": 0.04896574467420578, "reward_total_composite_mean": 0.4079222083091736, "reward_total_composite_std": 0.1708296537399292, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1671.0} {"timestamp_utc": "2026-04-11T23:02:09Z", "mode": "train", "global_step": 839, "epoch": 0.033698839217576416, "loss": 0.0131, "grad_norm": 2.164283275604248, "learning_rate": 7.460606060606061e-06, "num_tokens": 1863934.0, "completions/mean_length": 112.125, "completions/min_length": 109.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.125, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.9943764209747314, "rewards/meter/std": 0.0010588886216282845, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.795501172542572, "rewards/total_composite/std": 0.0008471080800518394, "reward": 0.795501172542572, "reward_std": 0.0008471118635497987, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008892908692359924, "sampling/sampling_logp_difference/max": 0.7849429249763489, "sampling/importance_sampling_ratio/min": 0.4561457335948944, "sampling/importance_sampling_ratio/mean": 1.0025136470794678, "sampling/importance_sampling_ratio/max": 1.7390761375427246, "entropy": 0.047560357954353094, "clip_ratio/low_mean": 0.006605345057323575, "clip_ratio/low_min": 0.006605345057323575, "clip_ratio/high_mean": 0.0022935778833925724, "clip_ratio/high_max": 0.0022935778833925724, "clip_ratio/region_mean": 0.008898922940716147, "reward_total_mean": 0.795501172542572, "reward_meter_mean": 0.9943764209747314, "reward_meter_std": 0.0010588886216282845, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.795501172542572, "reward_total_composite_std": 0.0008471080800518394, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1672.0} {"timestamp_utc": "2026-04-11T23:02:14Z", "mode": "train", "global_step": 840, "epoch": 0.03373900469936137, "loss": 0.036, "grad_norm": 5.7492289543151855, "learning_rate": 7.4575757575757575e-06, "num_tokens": 1865375.0, "completions/mean_length": 37.125, "completions/min_length": 36.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.7395036816596985, "rewards/meter/std": 0.2953052818775177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7395036816596985, "rewards/total_composite/std": 0.2953052818775177, "reward": 0.7395036816596985, "reward_std": 0.2953052818775177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01924203522503376, "sampling/sampling_logp_difference/max": 0.40451645851135254, "sampling/importance_sampling_ratio/min": 0.6779606938362122, "sampling/importance_sampling_ratio/mean": 1.0085185766220093, "sampling/importance_sampling_ratio/max": 1.4985777139663696, "entropy": 0.11148730758577585, "clip_ratio/low_mean": 0.006410256493836641, "clip_ratio/low_min": 0.006410256493836641, "clip_ratio/high_mean": 0.010228979168459773, "clip_ratio/high_max": 0.010228979168459773, "clip_ratio/region_mean": 0.016639235662296414, "reward_total_mean": 0.7395036816596985, "reward_meter_mean": 0.7395036816596985, "reward_meter_std": 0.2953052818775177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7395036816596985, "reward_total_composite_std": 0.2953052818775177, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1673.0} {"timestamp_utc": "2026-04-11T23:02:19Z", "mode": "train", "global_step": 841, "epoch": 0.033779170181146324, "loss": -0.0141, "grad_norm": 3.028710126876831, "learning_rate": 7.454545454545456e-06, "num_tokens": 1867245.0, "completions/mean_length": 64.75, "completions/min_length": 64.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9528952240943909, "rewards/meter/std": 0.01815449446439743, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9528952240943909, "rewards/total_composite/std": 0.01815449446439743, "reward": 0.9528952240943909, "reward_std": 0.018154479563236237, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01176523882895708, "sampling/sampling_logp_difference/max": 0.8644721508026123, "sampling/importance_sampling_ratio/min": 0.5180370807647705, "sampling/importance_sampling_ratio/mean": 1.0040351152420044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.045040544122457504, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.005771921598352492, "reward_total_mean": 0.9528952240943909, "reward_meter_mean": 0.9528952240943909, "reward_meter_std": 0.01815449446439743, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9528952240943909, "reward_total_composite_std": 0.01815449446439743, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1674.0} {"timestamp_utc": "2026-04-11T23:02:24Z", "mode": "train", "global_step": 842, "epoch": 0.03381933566293128, "loss": 0.0297, "grad_norm": 9.17272663116455, "learning_rate": 7.451515151515152e-06, "num_tokens": 1868749.0, "completions/mean_length": 37.0, "completions/min_length": 36.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.700322687625885, "rewards/meter/std": 0.3828321695327759, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.700322687625885, "rewards/total_composite/std": 0.3828321695327759, "reward": 0.700322687625885, "reward_std": 0.3828321397304535, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011855069547891617, "sampling/sampling_logp_difference/max": 0.964850664138794, "sampling/importance_sampling_ratio/min": 0.6239942312240601, "sampling/importance_sampling_ratio/mean": 1.0074673891067505, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04681507125496864, "clip_ratio/low_mean": 0.009699730202555656, "clip_ratio/low_min": 0.009699730202555656, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.009699730202555656, "reward_total_mean": 0.700322687625885, "reward_meter_mean": 0.700322687625885, "reward_meter_std": 0.3828321695327759, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.700322687625885, "reward_total_composite_std": 0.3828321695327759, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1675.0} {"timestamp_utc": "2026-04-11T23:02:33Z", "mode": "train", "global_step": 843, "epoch": 0.03385950114471623, "loss": 0.0211, "grad_norm": 2.23907470703125, "learning_rate": 7.448484848484849e-06, "num_tokens": 1874562.0, "completions/mean_length": 452.625, "completions/min_length": 426.0, "completions/max_length": 461.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 452.625, "completions/min_terminated_length": 426.0, "completions/max_terminated_length": 461.0, "rewards/meter/mean": 0.9962563514709473, "rewards/meter/std": 0.001195572316646576, "rewards/count_adherence/mean": 0.9270833730697632, "rewards/count_adherence/std": 0.029462777078151703, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5606521368026733, "rewards/repeat_penalty/std": 0.0018446200992912054, "rewards/total_composite/mean": 0.5178769826889038, "rewards/total_composite/std": 0.01842796988785267, "reward": 0.5178769826889038, "reward_std": 0.018427973613142967, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004932690877467394, "sampling/sampling_logp_difference/max": 4.1077117919921875, "sampling/importance_sampling_ratio/min": 0.01644536294043064, "sampling/importance_sampling_ratio/mean": 0.999667763710022, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.01467959355795756, "clip_ratio/low_mean": 0.0011018531513400376, "clip_ratio/low_min": 0.0011018531513400376, "clip_ratio/high_mean": 0.0002934272342827171, "clip_ratio/high_max": 0.0002934272342827171, "clip_ratio/region_mean": 0.0013952803856227547, "reward_total_mean": 0.5178769826889038, "reward_meter_mean": 0.9962563514709473, "reward_meter_std": 0.001195572316646576, "reward_count_adherence_mean": 0.9270833730697632, "reward_count_adherence_std": 0.029462777078151703, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5606521368026733, "reward_repeat_penalty_std": 0.0018446200992912054, "reward_total_composite_mean": 0.5178769826889038, "reward_total_composite_std": 0.01842796988785267, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1676.0} {"timestamp_utc": "2026-04-11T23:02:39Z", "mode": "train", "global_step": 844, "epoch": 0.033899666626501186, "loss": -0.0116, "grad_norm": 2.640986919403076, "learning_rate": 7.445454545454546e-06, "num_tokens": 1877374.0, "completions/mean_length": 149.5, "completions/min_length": 144.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.5, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.5467606782913208, "rewards/meter/std": 0.4782540798187256, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.3916763961315155, "rewards/total_composite/std": 0.34031400084495544, "reward": 0.3916763961315155, "reward_std": 0.34031397104263306, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019425617530941963, "sampling/sampling_logp_difference/max": 3.396358013153076, "sampling/importance_sampling_ratio/min": 0.03349503502249718, "sampling/importance_sampling_ratio/mean": 1.0008183717727661, "sampling/importance_sampling_ratio/max": 1.7641745805740356, "entropy": 0.09837253391742706, "clip_ratio/low_mean": 0.011086393264122307, "clip_ratio/low_min": 0.011086393264122307, "clip_ratio/high_mean": 0.002457805967424065, "clip_ratio/high_max": 0.002457805967424065, "clip_ratio/region_mean": 0.013544199231546372, "reward_total_mean": 0.3916763961315155, "reward_meter_mean": 0.5467606782913208, "reward_meter_std": 0.4782540798187256, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.3916763961315155, "reward_total_composite_std": 0.34031400084495544, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1677.0} {"timestamp_utc": "2026-04-11T23:02:44Z", "mode": "train", "global_step": 845, "epoch": 0.03393983210828614, "loss": 0.0112, "grad_norm": 6.265551567077637, "learning_rate": 7.442424242424243e-06, "num_tokens": 1879205.0, "completions/mean_length": 71.875, "completions/min_length": 68.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.522850513458252, "rewards/meter/std": 0.4167693555355072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.5038976669311523, "rewards/total_composite/std": 0.42369261384010315, "reward": 0.5038976669311523, "reward_std": 0.42369258403778076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03976872190833092, "sampling/sampling_logp_difference/max": 1.2724390029907227, "sampling/importance_sampling_ratio/min": 0.2801474928855896, "sampling/importance_sampling_ratio/mean": 1.0023622512817383, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16639000456780195, "clip_ratio/low_mean": 0.01767799479421228, "clip_ratio/low_min": 0.01767799479421228, "clip_ratio/high_mean": 0.017289764247834682, "clip_ratio/high_max": 0.017289764247834682, "clip_ratio/region_mean": 0.034967759042046964, "reward_total_mean": 0.5038976669311523, "reward_meter_mean": 0.522850513458252, "reward_meter_std": 0.4167693555355072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.5038976669311523, "reward_total_composite_std": 0.42369261384010315, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1678.0} {"timestamp_utc": "2026-04-11T23:02:52Z", "mode": "train", "global_step": 846, "epoch": 0.033979997590071094, "loss": -0.0053, "grad_norm": 4.094456672668457, "learning_rate": 7.439393939393939e-06, "num_tokens": 1883732.0, "completions/mean_length": 321.875, "completions/min_length": 309.0, "completions/max_length": 336.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 321.875, "completions/min_terminated_length": 309.0, "completions/max_terminated_length": 336.0, "rewards/meter/mean": 0.6540272831916809, "rewards/meter/std": 0.39356982707977295, "rewards/count_adherence/mean": 0.9861111044883728, "rewards/count_adherence/std": 0.03928370773792267, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4677932560443878, "rewards/repeat_penalty/std": 0.22017233073711395, "rewards/total_composite/mean": 0.3497573733329773, "rewards/total_composite/std": 0.2734247148036957, "reward": 0.3497573733329773, "reward_std": 0.2734247148036957, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007702399045228958, "sampling/sampling_logp_difference/max": 1.841371774673462, "sampling/importance_sampling_ratio/min": 0.15859971940517426, "sampling/importance_sampling_ratio/mean": 0.9985901713371277, "sampling/importance_sampling_ratio/max": 1.768311619758606, "entropy": 0.03188518015667796, "clip_ratio/low_mean": 0.0011160714784637094, "clip_ratio/low_min": 0.0011160714784637094, "clip_ratio/high_mean": 0.0058011687360703945, "clip_ratio/high_max": 0.0058011687360703945, "clip_ratio/region_mean": 0.006917240214534104, "reward_total_mean": 0.3497573733329773, "reward_meter_mean": 0.6540272831916809, "reward_meter_std": 0.39356982707977295, "reward_count_adherence_mean": 0.9861111044883728, "reward_count_adherence_std": 0.03928370773792267, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4677932560443878, "reward_repeat_penalty_std": 0.22017233073711395, "reward_total_composite_mean": 0.3497573733329773, "reward_total_composite_std": 0.2734247148036957, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1679.0} {"timestamp_utc": "2026-04-11T23:02:57Z", "mode": "train", "global_step": 847, "epoch": 0.03402016307185605, "loss": 0.045, "grad_norm": 6.007223606109619, "learning_rate": 7.4363636363636375e-06, "num_tokens": 1885713.0, "completions/mean_length": 72.625, "completions/min_length": 68.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.625, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.26388847827911377, "rewards/meter/std": 0.3744986653327942, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.2634860873222351, "rewards/total_composite/std": 0.3748124837875366, "reward": 0.2634860873222351, "reward_std": 0.3748124837875366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04990806803107262, "sampling/sampling_logp_difference/max": 1.3789244890213013, "sampling/importance_sampling_ratio/min": 0.2518492639064789, "sampling/importance_sampling_ratio/mean": 1.007526159286499, "sampling/importance_sampling_ratio/max": 1.518433690071106, "entropy": 0.2849195022135973, "clip_ratio/low_mean": 0.02969917980954051, "clip_ratio/low_min": 0.02969917980954051, "clip_ratio/high_mean": 0.019930581096559763, "clip_ratio/high_max": 0.019930581096559763, "clip_ratio/region_mean": 0.04962976090610027, "reward_total_mean": 0.2634860873222351, "reward_meter_mean": 0.26388847827911377, "reward_meter_std": 0.3744986653327942, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.2634860873222351, "reward_total_composite_std": 0.3748124837875366, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1680.0} {"timestamp_utc": "2026-04-11T23:03:02Z", "mode": "train", "global_step": 848, "epoch": 0.034060328553641, "loss": 0.0001, "grad_norm": 2.973163366317749, "learning_rate": 7.433333333333334e-06, "num_tokens": 1888010.0, "completions/mean_length": 106.125, "completions/min_length": 100.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.125, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.6358721256256104, "rewards/meter/std": 0.32246294617652893, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.5016140341758728, "rewards/total_composite/std": 0.2674255073070526, "reward": 0.5016140341758728, "reward_std": 0.2674255073070526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022296659648418427, "sampling/sampling_logp_difference/max": 1.8602898120880127, "sampling/importance_sampling_ratio/min": 0.1556275188922882, "sampling/importance_sampling_ratio/mean": 0.9971781969070435, "sampling/importance_sampling_ratio/max": 1.6591869592666626, "entropy": 0.08717937674373388, "clip_ratio/low_mean": 0.005942982388660312, "clip_ratio/low_min": 0.005942982388660312, "clip_ratio/high_mean": 0.013045326224528253, "clip_ratio/high_max": 0.013045326224528253, "clip_ratio/region_mean": 0.018988308613188565, "reward_total_mean": 0.5016140341758728, "reward_meter_mean": 0.6358721256256104, "reward_meter_std": 0.32246294617652893, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.5016140341758728, "reward_total_composite_std": 0.2674255073070526, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1681.0} {"timestamp_utc": "2026-04-11T23:03:10Z", "mode": "train", "global_step": 849, "epoch": 0.034100494035425956, "loss": 0.001, "grad_norm": 0.23050236701965332, "learning_rate": 7.430303030303031e-06, "num_tokens": 1892024.0, "completions/mean_length": 283.75, "completions/min_length": 283.0, "completions/max_length": 284.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 283.75, "completions/min_terminated_length": 283.0, "completions/max_terminated_length": 284.0, "rewards/meter/mean": 0.9965326189994812, "rewards/meter/std": 0.00021112659305799752, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6000000238418579, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5979195833206177, "rewards/total_composite/std": 0.00012666928523685783, "reward": 0.5979195833206177, "reward_std": 0.000126663115224801, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0016167466528713703, "sampling/sampling_logp_difference/max": 0.7609295845031738, "sampling/importance_sampling_ratio/min": 0.4672318696975708, "sampling/importance_sampling_ratio/mean": 1.0001165866851807, "sampling/importance_sampling_ratio/max": 1.2486354112625122, "entropy": 0.010985135799273849, "clip_ratio/low_mean": 0.00044014083687216043, "clip_ratio/low_min": 0.00044014083687216043, "clip_ratio/high_mean": 0.00044169611646793783, "clip_ratio/high_max": 0.00044169611646793783, "clip_ratio/region_mean": 0.0008818369533400983, "reward_total_mean": 0.5979195833206177, "reward_meter_mean": 0.9965326189994812, "reward_meter_std": 0.00021112659305799752, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6000000238418579, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5979195833206177, "reward_total_composite_std": 0.00012666928523685783, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1682.0} {"timestamp_utc": "2026-04-11T23:03:15Z", "mode": "train", "global_step": 850, "epoch": 0.03414065951721091, "loss": -0.0004, "grad_norm": 7.60121488571167, "learning_rate": 7.4272727272727275e-06, "num_tokens": 1893564.0, "completions/mean_length": 35.5, "completions/min_length": 33.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9917958974838257, "rewards/meter/std": 0.008829712867736816, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8676612377166748, "rewards/total_composite/std": 0.35069888830184937, "reward": 0.8676612377166748, "reward_std": 0.350698858499527, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04963725060224533, "sampling/sampling_logp_difference/max": 3.3897039890289307, "sampling/importance_sampling_ratio/min": 0.03371865674853325, "sampling/importance_sampling_ratio/mean": 1.0052940845489502, "sampling/importance_sampling_ratio/max": 1.9962736368179321, "entropy": 0.15574337635189295, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/high_mean": 0.031166881788522005, "clip_ratio/high_max": 0.031166881788522005, "clip_ratio/region_mean": 0.03830973897129297, "reward_total_mean": 0.8676612377166748, "reward_meter_mean": 0.9917958974838257, "reward_meter_std": 0.008829712867736816, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8676612377166748, "reward_total_composite_std": 0.35069888830184937, "run_id": "shaer_grpo_20260411_223409", "run_sequence_index": 1, "_plot_step": 1683.0}